import json, re, glob, os

KO = '/data/news/json/ko-0926data.json'
ko = json.load(open(KO, encoding='utf-8'))
SECS = ['brand_hotspots','vehicle_hotspots','social_hotspots',
        'hyundai_buzz_topics_domestic','hyundai_buzz_topics_international']

# 1) 修 brand 字段：아이토(AITO) → AITO(问界)
n = 0
for s in SECS:
    for e in ko.get(s, []):
        b = str(e.get('brand','') or '')
        if '아이토' in b or re.search(r'[\uac00-\ud7af]', b):
            print(f'  [{s}] brand: {b!r} → AITO(问界)')
            e['brand'] = 'AITO(问界)'
            n += 1
json.dump(ko, open(KO, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
print(f'brand 修复 {n} 处')
print()

# 2) 严格回归检测：历史形如 Word(中文) 才算「历史已注解」
hist = [f for f in sorted(glob.glob('/data/news/json/ko-0*.json'))
        if os.path.basename(f) != 'ko-0926data.json']
hist_txt = {os.path.basename(f): open(f, encoding='utf-8').read() for f in hist}

pat_lat = re.compile(r'[A-Za-z][A-Za-z0-9\.\- ]{2,30}')
seen = {}
for s in SECS:
    for e in ko.get(s, []):
        for f in ['title','summary','focus_point','brand']:
            t = str(e.get(f,'') or '')
            for m in pat_lat.finditer(t):
                w = m.group(0).strip()
                if w and not t[m.end():m.end()+12].startswith('('):
                    seen[w] = seen.get(w, 0) + 1

zh_pat = re.compile(r'[\u4e00-\u9fff]')
print('=== 严格回归检测：历史存在 Word(中文) 形态 ===')
reg = []
for w, c in sorted(seen.items(), key=lambda x: -x[1]):
    prec = []
    for fn, txt in hist_txt.items():
        for m in re.finditer(re.escape(w) + r'\(([^()]{0,20})\)', txt):
            if zh_pat.search(m.group(1)):
                prec.append(f'{fn}: {m.group(0)[:26]}')
                break
    if prec:
        reg.append((w, prec[:2]))
        print(f'  ⚠️ {w} ×{c} ← {prec[0]}')
    else:
        print(f'  ✅ {w} ×{c}')
print()
print('严格口径下回归疑似:', len(reg), '个')
