import json, re

KO = '/data/news/json/ko-0927data.json'
ko = json.load(open(KO, encoding='utf-8'))
SECS = ['brand_hotspots','vehicle_hotspots','social_hotspots',
        'hyundai_buzz_topics_domestic','hyundai_buzz_topics_international']

# ① 嵌套标注 + 韩文车型名 → Santa Fe(胜达)（先例 ko-0803/0810；现代系车型保持英文）
# ② 코나(Kona) → Kona（与同条 title/summary 一致，现代系车型保持英文）
# ③ Dify 漏译中文「授权」→ 라이선스
LITERAL = [
    ('싼타페(Santa Fe(胜达))', 'Santa Fe(胜达)'),
    ('싼타페(Santa Fe)', 'Santa Fe(胜达)'),
    ('코나(Kona)', 'Kona'),
    ('授权', '라이선스'),
]

n = 0
for s in SECS:
    for i, e in enumerate(ko.get(s, [])):
        for f in ['title', 'summary', 'focus_point', 'brand']:
            v = e.get(f)
            if not isinstance(v, str):
                continue
            new = v
            for a, b in LITERAL:
                new = new.replace(a, b)
            new = re.sub(r'(?<=[\uac00-\ud7af])라이선스', ' 라이선스', new)  # 补空格
            if new != v:
                print(f'  [{s}][{i}].{f}')
                print(f'     改前: {v[:100]}')
                print(f'     改后: {new[:100]}')
                e[f] = new
                n += 1

json.dump(ko, open(KO, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)
print(f'\n共修复 {n} 处')

# 复核：韩文残留 / 嵌套 / 中文残留
print()
pat_ko_en = re.compile(r'[\uac00-\ud7af]{2,10}\([A-Za-z][A-Za-z ]{1,30}\)')
pat_nest = re.compile(r'\([^()]*\([^()]*\)[^()]*\)')
for s in SECS:
    for i, e in enumerate(ko.get(s, [])):
        for f in ['title','summary','focus_point','brand']:
            t = str(e.get(f,'') or '')
            m1 = pat_ko_en.search(t); m2 = pat_nest.search(t)
            m3 = re.search(r'[\u4e00-\u9fff]', re.sub(r'\([^()]*\)','',t))
            if m1 or m2 or m3:
                print(f'  ⚠️ [{s}][{i}].{f}: ko_en={m1.group(0) if m1 else None} nest={m2.group(0) if m2 else None} cn={m3.group(0) if m3 else None}')
print('  复核完成')
