#!/usr/bin/env python3
"""手动补全 yes_data 中缺失/不合格的 summary 和 focus_point（用可用 DeepSeek key）"""
import json, re, urllib.request, os, sys

KEY = 'sk-00d69ca50b124474b02236121d4cf147'
API = 'https://api.deepseek.com/v1/chat/completions'
TARGET = '/data/news/json/yes_data/0820data.json'

def ds(prompt, max_tokens=200, timeout=60):
    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3, "max_tokens": max_tokens
    }).encode()
    req = urllib.request.Request(API, data=payload,
        headers={'Authorization': f'Bearer {KEY}', 'Content-Type': 'application/json'})
    resp = urllib.request.urlopen(req, timeout=timeout)
    return json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()

def fetch_content(url):
    """尝试 r.jina.ai 抓正文"""
    try:
        req = urllib.request.Request(f"https://r.jina.ai/{url}", headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=20)
        return resp.read().decode('utf-8')[:2500]
    except Exception:
        return ''

def gen_summary(title, content):
    if len(content.strip()) < 100:
        prompt = f'基于标题"{title}"生成summary（仅输出JSON）：\n只输出一个字段：summary，不超过40个汉字，客观概括核心事件。\n{{"summary": ""}}'
    else:
        prompt = f'阅读以下文章内容，生成summary（仅输出JSON）：\n只输出一个字段：summary，不超过40个汉字，包含核心事件和关键信息。\n\n【文章内容】\n{content}\n\n{{"summary": ""}}'
    try:
        reply = ds(prompt)
        m = re.search(r'"summary"\s*:\s*"([^"]+)"', reply)
        if m:
            return m.group(1)
    except Exception as e:
        print(f"    summary生成失败: {str(e)[:40]}")
    return ''

def gen_focus(title, summary):
    fp_title = title
    fp_summary = summary or title
    prompt = ('根据以下新闻，为现代汽车写一条策略启示（focus_point）。要求：不超过35个汉字，措辞柔和，用"可借鉴""可参考""不妨""建议"等友好语气，避免生硬的"应"。只输出JSON：{"focus_point": ""}\n\n标题：' + fp_title + '\n摘要：' + fp_summary)
    try:
        reply = ds(prompt)
        m = re.search(r'"focus_point"\s*:\s*"([^"]+)"', reply)
        if m:
            return m.group(1)
    except Exception as e:
        print(f"    focus_point生成失败: {str(e)[:40]}")
    return ''

d = json.load(open(TARGET))
fixed_s, fixed_f = 0, 0
for s in d:
    name = s.get('name')
    for i, it in enumerate(s.get('list', [])):
        t = it.get('title', '')
        sm = it.get('summary', '')
        fp = it.get('focus_point', '')
        sm_bad = (not sm) or len(sm) > 40 or (sm and not sm.endswith('。')) or sm == t
        fp_bad = (not fp) or len(fp) > 35
        if not (sm_bad or fp_bad):
            continue
        print(f"  [{name}][{i}] {t[:35]}")
        if sm_bad:
            content = fetch_content(it.get('source_url', ''))
            new_sm = gen_summary(t, content)
            if new_sm:
                it['summary'] = new_sm[:40] if not new_sm.endswith('。') else new_sm[:41]
                if not it['summary'].endswith('。'):
                    it['summary'] += '。'
                sm = it['summary']
                fixed_s += 1
                print(f"    ✅ summary: {sm[:40]}")
            else:
                print(f"    ⚠️ summary 仍空")
        if fp_bad:
            new_fp = gen_focus(t, sm)
            if new_fp:
                it['focus_point'] = new_fp[:35]
                fixed_f += 1
                print(f"    ✅ focus: {new_fp[:35]}")
            else:
                print(f"    ⚠️ focus 仍空")

json.dump(d, open(TARGET, 'w'), ensure_ascii=False, indent=2)
print(f"\n✅ 补全完成: summary {fixed_s} 条, focus_point {fixed_f} 条")
