import json

P = '/data/news/json/0926data.json'
d = json.load(open(P, encoding='utf-8'))

BASE = ['title','brand','summary','focus_point','thumb','source_url',
        'publish_time','platform']
SHAPE = {
    'brand_hotspots': BASE,
    'vehicle_hotspots': BASE,
    'social_hotspots': BASE + ['heat_score'],
    'hyundai_buzz_topics_domestic': BASE + ['icon'],
    'hyundai_buzz_topics_international': BASE + ['icon'],
}
DEFAULTS = {'icon': False, 'heat_score': 0}

n_fill, n_drop = 0, 0
for s, order in SHAPE.items():
    new_list = []
    for e in d.get(s, []):
        for k in order:
            if k not in e:
                e[k] = DEFAULTS.get(k, '')
                n_fill += 1
        for k in list(e.keys()):
            if k not in order:            # model / yesorno / 其他杂项
                del e[k]
                n_drop += 1
        new_list.append({k: e[k] for k in order})   # 规范化字段顺序
    d[s] = new_list

json.dump(d, open(P, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)

print(f'补回缺失核心字段: {n_fill} 处 | 删除非核心字段: {n_drop} 处')
print()
for s in SHAPE:
    lst = d.get(s, [])
    keys = set()
    for e in lst: keys |= set(e.keys())
    print(f'  {s}: {len(lst)} 条 | {sorted(keys)}')
print()
print('social[0].brand =', repr(d['social_hotspots'][0].get('brand')))
print('hyundai_dom[0].thumb =', repr(d['hyundai_buzz_topics_domestic'][0].get('thumb')))
print('hyundai_dom[0].icon =', repr(d['hyundai_buzz_topics_domestic'][0].get('icon')))
print('social[0].heat_score =', repr(d['social_hotspots'][0].get('heat_score')))
print('条数:', [len(d.get(s, [])) for s in SHAPE],
      '合计', sum(len(d.get(s, [])) for s in SHAPE))
