import json, os, re
from PIL import Image

P = '/data/news/json/0926data.json'
IMG = '/data/news/images/0926/'
d = json.load(open(P, encoding='utf-8'))
SECS = ['brand_hotspots','vehicle_hotspots','social_hotspots',
        'hyundai_buzz_topics_domestic','hyundai_buzz_topics_international']
SHAPE = {
    'brand_hotspots': ['title','brand','summary','focus_point','thumb','source_url','publish_time','platform'],
    'vehicle_hotspots': ['title','brand','summary','focus_point','thumb','source_url','publish_time','platform'],
    'social_hotspots': ['title','brand','summary','focus_point','thumb','source_url','publish_time','platform','heat_score'],
    'hyundai_buzz_topics_domestic': ['title','brand','summary','focus_point','thumb','source_url','publish_time','platform','icon'],
    'hyundai_buzz_topics_international': ['title','brand','summary','focus_point','thumb','source_url','publish_time','platform','icon'],
}
ok = lambda c: '✅' if c else '❌'

def ncontent(s): return len(re.sub(r'[^\w]', '', str(s), flags=re.UNICODE))

print('=== 1. 条数 ===')
c = [len(d.get(s, [])) for s in SECS]
print(f"  {ok(c == [5,5,5,3,3])} {c} 合计 {sum(c)}")

print('=== 2. 字段集 ===')
for s in SECS:
    for e in d[s]:
        extra = set(e) - set(SHAPE[s]); miss = set(SHAPE[s]) - set(e)
        if extra or miss:
            print(f'  ❌ [{s}] 多={extra} 缺={miss}')
print('  ✅ 全部字段集正确' if all(set(e)==set(SHAPE[s]) for s in SECS for e in d[s]) else '')

print('=== 3. report_date / hero / actions ===')
print(f"  report_date = {d.get('report_date')} {ok(d.get('report_date')=='2026-09-26')}")
print(f"  hero_summary {len(d.get('hero_summary',[]))} 条 {ok(len(d.get('hero_summary',[]))==3)} | 内容字 {[ncontent(h) for h in d.get('hero_summary',[])]}")
print(f"  strategy_actions {d.get('strategy_actions')} {ok(d.get('strategy_actions')==[])} (周五→应=[])")

print('=== 4. 标题/摘要长度 ===')
bt = [ (s,i,ncontent(e['title'])) for s in SECS for i,e in enumerate(d[s]) if ncontent(e['title'])>35 ]
bs = [ (s,i,ncontent(e['summary'])) for s in SECS for i,e in enumerate(d[s]) if ncontent(e['summary'])>45 ]
print(f"  {ok(not bt)} 标题>35字: {bt or '无'}")
print(f"  {ok(not bs)} 摘要>45字: {bs or '无'}")

print('=== 5. brand 齐（brand_hotspots/vehicle）===')
bad = [(s,i) for s in ['brand_hotspots','vehicle_hotspots'] for i,e in enumerate(d[s]) if not str(e.get('brand','')).strip()]
print(f"  {ok(not bad)} 空 brand: {bad or '无'}")

print('=== 6. icon / heat_score ===')
ic = sum(1 for s in ['hyundai_buzz_topics_domestic','hyundai_buzz_topics_international'] for e in d[s] if e.get('icon'))
hs = [e.get('heat_score') for e in d['social_hotspots']]
print(f"  {ok(ic==6)} hyundai icon = {ic}/6")
print(f"  {ok(all(isinstance(x,int) and x>0 for x in hs))} heat_score = {hs}")

print('=== 7. 配图 ===')
miss, nonjpeg, vert = [], [], []
n = 0
for s in ['brand_hotspots','vehicle_hotspots','social_hotspots']:
    for i,e in enumerate(d[s]):
        t = str(e.get('thumb',''))
        n += 1
        p = os.path.join('/data/news', t.lstrip('/'))
        if not t.startswith('/images/0926/') or not os.path.exists(p):
            miss.append((s,i,t)); continue
        im = Image.open(p)
        if im.format != 'JPEG': nonjpeg.append((s,i,im.format))
        if im.width <= im.height: vert.append((s,i,im.size))
        if os.path.getsize(p) < 20000: miss.append((s,i,f'{os.path.getsize(p)}B'))
print(f"  {ok(not miss)} 15 条 thumb 指向 images/0926 且存在: {miss or '全部 OK'}")
print(f"  {ok(not nonjpeg)} 真 JPEG: {nonjpeg or '全部 JPEG'}")
print(f"  {ok(not vert)} 横向: {vert or '全部横向'}")

print('=== 8. hyundai thumb（应为空，cron 负责本地化）===')
print('  ', [(s, repr(d[s][0].get('thumb'))) for s in SECS[3:]])
