#!/usr/bin/python3.12
"""精简 0817 超长 summary（hyundai 35-40字，其他 ≤45字）"""
import json, urllib.request, time, re

API_URL = 'https://api.deepseek.com/v1/chat/completions'
KEY = 'sk-00d69ca50b124474b02236121d4cf147'

def call_deepseek(prompt, max_tokens=400):
    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": max_tokens
    }).encode('utf-8')
    req = urllib.request.Request(API_URL, data=payload, headers={
        'Content-Type': 'application/json',
        'Authorization': f'Bearer {KEY}'
    })
    for attempt in range(3):
        try:
            with urllib.request.urlopen(req, timeout=60) as r:
                return json.loads(r.read().decode('utf-8'))['choices'][0]['message']['content']
        except Exception as e:
            print(f"  ⚠️ 重试{attempt+1}: {e}")
            time.sleep(3)
    return None

def cnt(text):
    return len(re.sub(r'[；，。、：""''（）()\s]', '', text))

path = '/data/news/json/yes_data/0817data.json'
d = json.load(open(path))

# 需要精简的：hyundai >40字，其他 >45字
targets = []
for s in d:
    for i, it in enumerate(s.get('list', [])):
        sm = it.get('summary', '')
        if not sm:
            continue
        n = cnt(sm)
        limit = 40 if s.get('name', '').startswith('hyundai') else 45
        if n > limit:
            targets.append((s.get('name'), i, it.get('title', ''), sm, limit))

print(f"需要精简 {len(targets)} 条")
for sect_name, idx, title, sm, limit in targets:
    prompt = f"""请将下面这条新闻摘要精简到{limit}字以内（不含标点），要求：
1. 保留核心信息：谁+做了什么+关键数据
2. 去掉冗余修饰词，保留最有信息量的部分
3. 直接输出精简后的摘要文本，不要解释

原摘要：{sm}"""
    resp = call_deepseek(prompt)
    if resp:
        new_sm = resp.strip().replace('\n', '')
        for prefix in ['摘要：', '摘要:']:
            if new_sm.startswith(prefix):
                new_sm = new_sm[len(prefix):]
        n = cnt(new_sm)
        # 找到对应条目更新
        for s in d:
            if s.get('name') == sect_name and idx < len(s['list']):
                s['list'][idx]['summary'] = new_sm
                break
        status = "✅" if n <= limit else "⚠️仍超"
        print(f"  {status} {sect_name}[{idx}] {n}字 (限{limit}): {new_sm}")
    else:
        print(f"  ❌ {sect_name}[{idx}] 精简失败")
    time.sleep(1)

json.dump(d, open(path, 'w'), ensure_ascii=False, indent=2)
print("\n✅ 已保存")
