#!/usr/bin/env python3.12
# -*- coding: utf-8 -*-
"""修复 social summary + focus_point 补写（Phase 3 修复脚本）"""
import json, re, urllib.request, yaml, sys

PATH = '/data/news/json/yes_data/0821data.json'
d = json.load(open(PATH))

# ── 1. social summary + source_url/platform 修复 ──
fixes = {
    0: {'summary': '交通运输部印发《精品自驾旅游公路实施方案》，构建"1+N"精品自驾旅游公路体系，规划约6万公里"三环、四横、五纵"国家级自驾路线。'},
    1: {'source_url': 'https://www.chinanews.com.cn/cj/2026/08-21/10681531.shtml', 'platform': '中国新闻网',
        'summary': '财政部在国新办发布会介绍"十五五"财政政策：支出更多转向投资于人、支持国内消费，以零基预算改革优化结构，增强国内大循环内生动力。'},
    2: {'source_url': 'https://news.china.com/socialgd/10000169/20260806/49659271.html', 'platform': '中华网',
        'summary': '宇树与长鑫同属科创板硬科技IPO，但本金门槛、中签率与估值逻辑截然不同：长鑫是"小本金可观收益"肉签，宇树中签率仅其约1/15，属高本金高风险。'},
    4: {'summary': '马斯克在X平台表示AI领域中国是其最强大竞争对手，并转发15年前"终局之战将与中国巅峰较量"的预测推文。'},
}
for s in d:
    if s.get('name') == 'social_hotspots':
        for idx, it in enumerate(s.get('list', [])):
            if idx in fixes:
                for k, v in fixes[idx].items():
                    it[k] = v
                print(f'  ✅ social[{idx}] 已更新 {list(fixes[idx].keys())}')
        break

# ── 2. focus_point 补写（DeepSeek）──
# 读 key
key = None
try:
    with open('/root/.openclaw/openclaw.json') as f:
        key = json.load(f).get('api_key') or json.load(f).get('deepseek_api_key')
except Exception:
    pass
if not key:
    cfg = yaml.safe_load(open('/home/ubuntu/.hermes/config.yaml'))
    key = cfg['providers']['deepseek']['api_key']

DEEPSEEK_API_URL = 'https://api.deepseek.com/chat/completions'

def gen_focus_point(title, summary):
    prompt = ('根据以下新闻，为现代汽车写一条策略启示（focus_point）。要求：不超过35个汉字，措辞柔和，'
              '用"可借鉴""可参考""不妨""建议"等友好语气，避免生硬的"应"。只输出JSON：{"focus_point": ""}\n\n'
              f'标题：{title}\n摘要：{summary or title}')
    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 200
    }).encode()
    req = urllib.request.Request(DEEPSEEK_API_URL, data=payload,
        headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
    resp = urllib.request.urlopen(req, timeout=60)
    data = json.loads(resp.read().decode())
    content = data['choices'][0]['message']['content']
    m = re.search(r'"focus_point"\s*:\s*"([^"]+)"', content)
    if m:
        return m.group(1)
    return content.strip().strip('{}"')

need_fp = []
for s in d:
    name = s.get('name')
    if name in ('vehicle_hotspots', 'social_hotspots', 'hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international'):
        for idx, it in enumerate(s.get('list', [])):
            if not it.get('focus_point'):
                need_fp.append((name, idx, it))

print(f'\n需要补写 focus_point: {len(need_fp)} 条')
for name, idx, it in need_fp:
    try:
        fp = gen_focus_point(it.get('title', ''), it.get('summary', ''))
        it['focus_point'] = fp
        print(f'  ✅ {name}[{idx}] → {fp}')
    except Exception as e:
        print(f'  ❌ {name}[{idx}] 失败: {e}')

json.dump(d, open(PATH, 'w'), ensure_ascii=False, indent=2)
print('\n✅ 已保存')

# ── 3. 校验 ──
d2 = json.load(open(PATH))
empty_fp = []
empty_sum = []
for s in d2:
    for i, it in enumerate(s.get('list', [])):
        if not it.get('focus_point'):
            empty_fp.append(f'{s.get("name")}[{i}]')
        if s.get('name') == 'social_hotspots' and not it.get('summary'):
            empty_sum.append(f'social[{i}]')
print(f'剩余 focus_point 空: {len(empty_fp)} {empty_fp}')
print(f'剩余 social summary 空: {len(empty_sum)} {empty_sum}')
