#!/usr/bin/env python3.12
"""第五阶段：韩译后处理 12 项核验（可复用）

用法: python3.12 verify_ko_12.py [MMDD]      # 默认 MMDD+1 = 今天(运行日)+1
规则依据: references/translation-postprocessing-rules.md §4
"""
import json, re, sys, os, glob
from datetime import date, timedelta

MMDD = sys.argv[1] if len(sys.argv) > 1 else (date.today() + timedelta(days=1)).strftime('%m%d')
KO = f'/data/news/json/ko-{MMDD}data.json'
CN = f'/data/news/json/{MMDD}data.json'
SECS = ['brand_hotspots', 'vehicle_hotspots', 'social_hotspots',
        'hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international']

d = json.load(open(KO, encoding='utf-8'))
c = json.load(open(CN, encoding='utf-8'))
ok = lambda b: '✅' if b else '⛔'
res = []

def rep(n, name, passed, detail=''):
    res.append((n, name, passed, detail))
    print(f'  {ok(passed)} [{n:2d}] {name}' + (f' — {detail}' if detail else ''))

def cl(s): return len(re.sub(r'[^\w]', '', str(s or ''), flags=re.UNICODE))

print(f'=== 韩译 12 项核验: {KO} ===\n')

# ---------- 1 条数 ----------
cnt = [len(d.get(s, [])) for s in SECS]
rep(1, '条数 5/5/5/3/3', cnt == [5, 5, 5, 3, 3], str(cnt))

# ---------- 2 韩文(英文)残留 ----------
P2 = re.compile(r'[\uac00-\ud7af]{2,10}\([A-Za-z][A-Za-z ]{1,30}\)')
hits = []
for s in SECS:
    for i, e in enumerate(d.get(s, []), 1):
        for k, v in e.items():
            if isinstance(v, str) and P2.search(v):
                hits.append(f'{s}[{i}].{k}: {v[:60]}')
rep(2, '韩文(拉丁)残留 = 0', not hits, '; '.join(hits[:4]) if hits else '')

# ---------- 3 韩文品牌/车型漂移（排除 KEEP） ----------
KEEP_KO = {'북경현대', '제네시스', '현대자동차'}
KO_BRANDS = ['싼타페', '코나', '아반떼', '스타리아', '투싼', '쏘나타', '그랜저', '팰리세이드',
             '스포티지', '셀토스', '카니발', '아이오닉', '넥쏘', '모하비', '베뉴', '캐스퍼',
             '리오토', '란투', '기아', '비야디', '지리', '창안', '웨이', '아이토']
P3 = re.compile('|'.join(KO_BRANDS))
hits3 = []
for s in SECS:
    for i, e in enumerate(d.get(s, []), 1):
        for k, v in e.items():
            if not isinstance(v, str): continue
            for m in P3.finditer(v):
                tok = m.group(0)
                if tok in KEEP_KO: continue
                hits3.append(f'{s}[{i}].{k}: …{v[max(0,m.start()-8):m.end()+10]}…')
rep(3, '韩文品牌漂移 = 0（KEEP 已排除）', not hits3, '; '.join(hits3[:4]) if hits3 else '')

# ---------- 4 platform == 中文源 ----------
hits4 = []
for s in SECS:
    for i, (a, b) in enumerate(zip(d.get(s, []), c.get(s, [])), 1):
        if a.get('platform') != b.get('platform'):
            hits4.append(f'{s}[{i}] ko={a.get("platform")!r} cn={b.get("platform")!r}')
rep(4, 'platform == 中文源', not hits4, '; '.join(hits4[:4]) if hits4 else '')

# ---------- 5 icon 保留 ----------
hits5 = []
for s in SECS[3:]:
    for i, (a, b) in enumerate(zip(d.get(s, []), c.get(s, [])), 1):
        if bool(a.get('icon')) != bool(b.get('icon')):
            hits5.append(f'{s}[{i}] ko={a.get("icon")} cn={b.get("icon")}')
rep(5, 'hyundai icon 保留', not hits5 and all('icon' in e for s in SECS[3:] for e in d[s]),
    '; '.join(hits5[:4]) if hits5 else f"{sum(1 for s in SECS[3:] for e in d[s] if e.get('icon'))}/6")

# ---------- 6 heat_score / thumb 保留（social） ----------
hits6 = []
for i, (a, b) in enumerate(zip(d['social_hotspots'], c['social_hotspots']), 1):
    if a.get('heat_score') != b.get('heat_score'):
        hits6.append(f'social[{i}] ko={a.get("heat_score")} cn={b.get("heat_score")}')
    if str(a.get('thumb', '')).startswith('/') and a.get('thumb') != b.get('thumb'):
        hits6.append(f'social[{i}] thumb 变了 {a.get("thumb")} vs {b.get("thumb")}')
rep(6, 'social heat_score / thumb 保留', not hits6, '; '.join(hits6[:4]) if hits6 else '')

# ---------- 7 thumb/source_url/publish_time 未被翻译 ----------
hits7 = []
KR = re.compile(r'[\uac00-\ud7af]')
for s in SECS:
    for i, e in enumerate(d.get(s, []), 1):
        for k in ('thumb', 'source_url', 'publish_time'):
            v = str(e.get(k, ''))
            if v.startswith('/images/') or v.startswith('http'):
                if KR.search(v): hits7.append(f'{s}[{i}].{k}={v[:50]}')
            elif k == 'publish_time' and KR.search(v):
                hits7.append(f'{s}[{i}].publish_time={v[:30]}')
rep(7, 'thumb/source_url/publish_time 未被翻译', not hits7, '; '.join(hits7[:4]) if hits7 else '')

# ---------- 8 无空字段（hyundai thumb 例外） ----------
EMPTY_SKIP = {('hyundai_buzz_topics_domestic', 'thumb'), ('hyundai_buzz_topics_international', 'thumb')}
hits8 = []
for s in SECS:
    for i, e in enumerate(d.get(s, []), 1):
        for k in ('title', 'summary', 'focus_point', 'thumb', 'source_url', 'publish_time', 'platform'):
            if (s, k) in EMPTY_SKIP: continue
            if not str(e.get(k, '')).strip():
                hits8.append(f'{s}[{i}].{k}')
rep(8, '无空字段（hyundai thumb 例外）', not hits8, '; '.join(hits8[:6]) if hits8 else '')

# ---------- 9 括号外中文残留（媒体名白名单） ----------
MEDIA = ['网通社', '满电', '汽车之家', 'IT之家', '易车', '腾讯新闻', '网易', '新浪', '搜狐',
         '百度', '中关村在线', '数英网', '品牌星球', '广告门', '钛媒体', '雷锋网', '澎湃新闻',
         '光明网', '中国青年报', '央视新闻', '第一财经', '界面新闻', '虎嗅', '懂车帝', '快科技',
         '驱动之家', '太平洋汽车', '爱卡汽车', '盖世汽车', '电车之家', '车质网', '亿邦动力',
         '财经', '日报', '周刊', '商报', '时报', '网', '号', '车家号', '百家号', 'SocialBeta',
         'CLauto', 'IT时代', '满电', '车市', '汽车', '新出行', '每经', '新华', '人民',
         '同花顺', '东方财富', '雪球', '股吧', '新浪财经', '南方', '封面', '九派', '极目']
P9 = re.compile(r'[\u4e00-\u9fff]{2,}')
hits9 = []
for s in SECS:
    for i, e in enumerate(d.get(s, []), 1):
        for k in ('title', 'summary', 'focus_point'):
            v = str(e.get(k, ''))
            stripped = re.sub(r'\([^()]*\)', '', v)   # 去掉标注
            stripped = re.sub(r'[（）]', '', stripped)
            remain = P9.findall(stripped)
            remain = [w for w in remain if not any(w in m for m in MEDIA)]
            if k == 'focus_point':
                # focus_point 按 §0.3 不带标注 → 中文残留判定放宽（只报 ≥4 字的短语）
                remain = [w for w in remain if len(w) >= 4]
            if remain:
                hits9.append(f'{s}[{i}].{k}: {remain[:4]}')
rep(9, '括号外中文残留 = 0（媒体名白名单）', not hits9, '; '.join(hits9[:4]) if hits9 else '')

# ---------- 10 双标注 / 括号嵌套 ----------
P10a = re.compile(r'\([^()]*\([^()]*\)[^()]*\)')      # 嵌套
P10b = re.compile(r'\(([^()]*)\)\s*\(([^()]*)\)')     # 相邻双标注
hits10 = []
for s in SECS:
    for i, e in enumerate(d.get(s, []), 1):
        for k, v in e.items():
            if not isinstance(v, str): continue
            for m in P10a.finditer(v): hits10.append(f'嵌套 {s}[{i}].{k}: {m.group(0)[:50]}')
            for m in P10b.finditer(v): hits10.append(f'相邻 {s}[{i}].{k}: {m.group(0)[:50]}')
rep(10, '双标注/括号嵌套 = 0（嵌套+相邻两个正则）', not hits10, '; '.join(hits10[:4]) if hits10 else '')

# ---------- 11 hero_summary / strategy_actions ----------
hn = len(d.get('hero_summary', []))
sa = len(d.get('strategy_actions', []) or [])
run_dow = date.today().weekday()          # 0=Mon ... 6=Sun
rd = d.get('report_date', '')
try:
    rd_dow = date.fromisoformat(rd).weekday()
except Exception:
    rd_dow = None
is_sun = (run_dow == 6) or (rd_dow == 6)
expect_sa = 3 if is_sun else 0
rep(11, f'hero=3 条 / strategy_actions 按周日规则(应={expect_sa})', hn == 3 and sa == expect_sa,
    f'hero={hn}, sa={sa}, 运行日={"周日" if run_dow==6 else "非周日"}, report_date={rd}')

# ---------- 12 幂等（由外部单独验证，这里只做提示） ----------
print('  ℹ️  [12] 幂等性需单独验证：translate_replace.py 第二遍须 0 处；restore_platform_names.py 第二遍须 0 处')

# ---------- 汇总 ----------
fails = [r for r in res if not r[2]]
print(f'\n{"="*60}')
print(f'结果: {len(res)-len(fails)}/{len(res)} 通过' + ('  →  ✅ ALL PASS' if not fails else ''))
if fails:
    print('⛔ 未通过项:')
    for n, name, _, det in fails:
        print(f'   [{n}] {name} — {det}')
sys.exit(1 if fails else 0)
