"""重新处理 vehicle_hotspots: Stage 1 粗筛取 top 15 → Stage 2 DeepSeek
其余 39 条直接设为 reject
"""

import json, os, urllib.request
from datetime import datetime

DEEPSEEK_API_URL = 'https://api.deepseek.com/v1/chat/completions'

KEYWORD_MAP = {
    '体育营销': ['世界杯', '奥运', '体育', '赛事', '运动员', '欧冠', '决赛', '球迷', '观赛', '比赛', '球场'],
    '情感营销': ['情感', '共鸣', '温度', '故事', '怀旧', '情怀', '温情', '暖心', '感动', '治愈'],
    '用户运营': ['用户', '私域', '社群', '会员', '粉丝', '圈层', '忠诚', '车主', '社区', '互动', '打卡'],
    '线下体验': ['体验', '场景', '快闪', '线下', '门店', '沉浸', '试驾', '到店', '工厂', '探访'],
    '本土化': ['本土化', '本土', '中国风', '国潮', '传统', '文化', '非遗', '国货', '中国'],
    'AI/数字化': ['AI', '人工智能', '数字化', '数据', '算力', '算法', '大模型', '智能系统', '芯片', '半导体'],
    '内容营销': ['短剧', '内容', '短视频', '直播', '共创', 'UGC', '短片', '视频', '广告', '种草'],
    '新车': ['上市', '发布', '预售', '亮相', '首发', '开启预售', '新车'],
    '电动化': ['电动', '纯电', 'EV', '续航', '充电', '新能源', '纯电动'],
    '混动': ['混动', 'PHEV', '增程', '轻混', '混动版', '插混'],
    '智能化': ['智能', '智驾', '自动驾驶', '座舱', '激光雷达', '乾崑', '天枢'],
    '价格/权益': ['售价', '万元', '万起', '补贴', '优惠', '限时', '福利', '降价', '置换', '金融', '低至'],
}

WEIGHTS = {
    '用户运营': 0.71, '线下体验': 0.44, '本土化': 0.59, 'AI/数字化': 0.28,
    '内容营销': 0.15, '新车': 0.99, '电动化': 0.74, '混动': 0.38,
    '智能化': 0.55, '价格/权益': 0.65,
}
TOTAL_W = sum(WEIGHTS.values())


def get_deepseek_key():
    cfg_path = '/root/.openclaw/openclaw.json'
    try:
        with open(cfg_path) as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except:
        return None


def stage1_score(text):
    text_lower = text.lower()
    hit_weight = 0.0
    hit_count = 0
    for dim, keywords in KEYWORD_MAP.items():
        w = WEIGHTS.get(dim, 0)
        for kw in keywords:
            if kw.lower() in text_lower:
                hit_weight += w
                hit_count += 1
                break
    raw = (hit_weight / TOTAL_W) * 100
    bonus = min(hit_count * 5, 20)
    score = min(100, raw + bonus)
    now = datetime.now()
    if (now.month == 6 and now.day >= 12) or (now.month == 7 and now.day <= 20):
        if any(kw in text_lower for kw in ['世界杯', '赛事', '观赛', '球迷', '足球']):
            score = min(100, score * 1.4)
    return score


def call_deepseek(items):
    key = get_deepseek_key()
    if not key:
        return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append(f"[{i}] 标题：{it.get('title', '')}")
        lines.append(f"    摘要：{it.get('summary', '')}")
    items_text = "\n".join(lines)

    prompt = f'''你是一位汽车行业热点分析师。请为以下{len(items)}条车型热点逐条判定综合价值评分。

评分标准（0-100）：
90-100：重磅新车上市/预售，或影响行业格局的重大事件
70-89：重要车型动态（新车发布、技术突破、价格调整等）
50-69：普通车型信息
0-49：与车型/汽车行业无关
注意：关注新车上市、价格变动、技术突破、竞品动态

请严格按照以下JSON格式输出，只输出JSON数组：
[
  {{"idx": 1, "score": 85}},
  ...
]

待评新闻：
{items_text}'''

    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 2000
    }).encode()

    req = urllib.request.Request(
        DEEPSEEK_API_URL, data=payload,
        headers={
            'Authorization': f'Bearer {key}',
            'Content-Type': 'application/json'
        }
    )
    try:
        resp = urllib.request.urlopen(req, timeout=60)
        result = json.loads(resp.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content'].strip()
        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            scores = json.loads(reply[json_start:json_end])
            return {s.get('idx'): max(0, min(100, int(s.get('score', 0)))) for s in scores}
        return None
    except Exception as e:
        print(f"  ❌ DeepSeek失败: {e}")
        return None


def main():
    path = '/data/news/json/origin_data/0615data.json'
    with open(path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    for section in data:
        if section.get('name') != 'vehicle_hotspots':
            continue

        items = section.get('list', [])
        total = len(items)
        print(f"车型热点共 {total} 条")

        # Stage 1
        print(f"\n▶ Stage 1 关键词评分...")
        scored = []
        for it in items:
            text = it.get('title', '') + ' ' + it.get('summary', '')
            s = stage1_score(text)
            scored.append((s, it))
        scored.sort(key=lambda x: x[0], reverse=True)

        top15 = scored[:15]
        rest = scored[15:]
        print(f"  前15名分数范围: {top15[-1][0]:.0f}-{top15[0][0]:.0f}")
        print(f"  其余 {len(rest)} 条直接设为 reject")

        for _, it in rest:
            it['yesorno'] = 'reject'

        # Stage 2
        print(f"\n▶ Stage 2 DeepSeek 评分（15条）...")
        top_items = [it for _, it in top15]
        for i, it in enumerate(top_items):
            print(f"  [{i+1}] {it.get('title','')[:45]}")

        score_map = call_deepseek(top_items)
        if score_map:
            for local_idx, it in enumerate(top_items):
                score = score_map.get(local_idx + 1, 50)
                if score >= 65:
                    it['yesorno'] = 'strong_select'
                elif score >= 45:
                    it['yesorno'] = 'select'
                elif score >= 25:
                    it['yesorno'] = 'backup'
                else:
                    it['yesorno'] = 'reject'
            print(f"  ✅ DeepSeek 返回成功")
        else:
            print(f"  ⚠️ DeepSeek 无返回，使用 Stage 1 分数")
            for s1, it in top15:
                if s1 >= 65:
                    it['yesorno'] = 'strong_select'
                elif s1 >= 45:
                    it['yesorno'] = 'select'
                elif s1 >= 25:
                    it['yesorno'] = 'backup'
                else:
                    it['yesorno'] = 'reject'

        counts = {'strong_select': 0, 'select': 0, 'backup': 0, 'reject': 0}
        for it in items:
            v = it.get('yesorno', '')
            if v in counts:
                counts[v] += 1
        print(f"\n▶ 最终分布:")
        for k, v in counts.items():
            print(f"  {k}: {v}条")

        break

    with open(path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print(f"\n✅ 已更新 {path}")


if __name__ == '__main__':
    main()
