import json

P = '/data/news/json/0926data.json'
d = json.load(open(P, encoding='utf-8'))

SECS = ['brand_hotspots','vehicle_hotspots','social_hotspots',
        'hyundai_buzz_topics_domestic','hyundai_buzz_topics_international']
CORE = ['title','brand','summary','focus_point','thumb','source_url',
        'publish_time','platform','icon','heat_score']

print('=== 清理前 ===')
for s in SECS:
    lst = d.get(s, [])
    keys = set()
    for e in lst: keys |= set(e.keys())
    print(f'  {s}: {len(lst)} 条 | 字段: {sorted(keys)}')

# 清理：只留 CORE，且删除全空字段
removed = {}
for s in SECS:
    for e in d.get(s, []):
        for k in list(e.keys()):
            if k not in CORE or (k != 'heat_score' and e[k] in ('', None, [])):
                removed.setdefault(k, 0)
                removed[k] += 1
                del e[k]

json.dump(d, open(P, 'w', encoding='utf-8'), ensure_ascii=False, indent=2)

print()
print('=== 清理后 ===')
for s in SECS:
    lst = d.get(s, [])
    keys = set()
    for e in lst: keys |= set(e.keys())
    print(f'  {s}: {len(lst)} 条 | 字段: {sorted(keys)}')
print()
print('删除的字段及次数:', removed or '无')
print('条数合计:', sum(len(d.get(s, [])) for s in SECS),
      [len(d.get(s, [])) for s in SECS])
print('icon 保留:', sum(1 for s in SECS for e in d.get(s, []) if 'icon' in e),
      '| heat_score 保留:', sum(1 for s in SECS for e in d.get(s, []) if 'heat_score' in e))
