"""重新处理 social_hotspots: Stage 1 粗筛取 top 30 → Stage 2 DeepSeek
其余 254 条直接设为 reject
"""

import json, os, re, urllib.request
from datetime import datetime

DEEPSEEK_API_URL = 'https://api.deepseek.com/v1/chat/completions'

KEYWORD_MAP = {
    '体育营销': ['世界杯', '奥运', '体育', '赛事', '运动员', '欧冠', '决赛', '球迷', '观赛', '比赛', '球场'],
    '用户运营': ['用户', '私域', '社群', '会员', '粉丝', '圈层', '忠诚', '车主', '社区', '互动', '打卡'],
    '线下体验': ['体验', '场景', '快闪', '线下', '门店', '沉浸', '试驾', '到店', '工厂', '探访'],
    '本土化': ['本土化', '本土', '中国风', '国潮', '传统', '文化', '非遗', '国货', '中国'],
    'AI/数字化': ['AI', '人工智能', '数字化', '数据', '算力', '算法', '大模型', '智能系统', '芯片', '半导体'],
    '智能化': ['智能', '智驾', '自动驾驶', '座舱', '激光雷达', '乾崑', '天枢'],
    '价格/权益': ['售价', '万元', '万起', '补贴', '优惠', '限时', '福利', '降价', '置换', '金融', '低至'],
    '消费/经济': ['消费', '经济', '就业', '收入', '物价', '房价', '补贴', '政策'],
    '出行/交通': ['出行', '交通', '自驾', '旅游', '通勤', '航空', '高铁', '地铁'],
}

WEIGHTS_SOCIAL = {
    '体育营销': 0.39, '用户运营': 0.10, '线下体验': 0.10, '本土化': 0.10,
    'AI/数字化': 0.20, '智能化': 0.10, '价格/权益': 0.08,
    '消费/经济': 0.30, '出行/交通': 0.25,
}
TOTAL_SOCIAL = sum(WEIGHTS_SOCIAL.values())


def get_deepseek_key():
    cfg_path = '/root/.openclaw/openclaw.json'
    try:
        with open(cfg_path) as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except:
        return None


def stage1_score(text):
    text_lower = text.lower()
    hit_weight = 0.0
    hit_count = 0
    for dim, keywords in KEYWORD_MAP.items():
        w = WEIGHTS_SOCIAL.get(dim, 0)
        for kw in keywords:
            if kw.lower() in text_lower:
                hit_weight += w
                hit_count += 1
                break
    raw = (hit_weight / TOTAL_SOCIAL) * 100
    bonus = min(hit_count * 5, 20)
    score = min(100, raw + bonus)
    now = datetime.now()
    if (now.month == 6 and now.day >= 12) or (now.month == 7 and now.day <= 20):
        if any(kw in text_lower for kw in ['世界杯', '赛事', '观赛', '球迷', '足球', '世界杯赞助', '赛场', '球队', '球星']):
            score = min(100, score * 1.4)
    return score


def call_deepseek(items):
    key = get_deepseek_key()
    if not key:
        return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append(f"[{i}] 标题：{it.get('title', '')}")
        lines.append(f"    摘要：{it.get('summary', '')}")
    items_text = "\n".join(lines)

    prompt = f'''你是一位汽车行业热点分析师。请为以下{len(items)}条社会热点逐条判定综合价值评分。

评分标准（0-100）：
90-100：与汽车消费、出行、消费趋势高度相关的全网热点
70-89：有影响力的社会话题，可结合汽车品牌借势
50-69：普通社会新闻，有一定参考价值
0-49：低价值或与汽车/消费场景无关

请严格按照以下JSON格式输出，只输出JSON数组：
[
  {{"idx": 1, "score": 85}},
  ...
]

待评新闻：
{items_text}'''

    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 2000
    }).encode()

    req = urllib.request.Request(
        DEEPSEEK_API_URL, data=payload,
        headers={
            'Authorization': f'Bearer {key}',
            'Content-Type': 'application/json'
        }
    )
    try:
        resp = urllib.request.urlopen(req, timeout=60)
        result = json.loads(resp.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content'].strip()
        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            scores = json.loads(reply[json_start:json_end])
            return {s.get('idx'): max(0, min(100, int(s.get('score', 0)))) for s in scores}
        return None
    except Exception as e:
        print(f"  ❌ DeepSeek失败: {e}")
        return None


def main():
    path = '/data/news/json/origin_data/0615data.json'
    with open(path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    for section in data:
        if section.get('name') != 'social_hotspots':
            continue

        items = section.get('list', [])
        total = len(items)
        print(f"社会热点共 {total} 条")

        # Stage 1: 关键词粗筛
        print(f"\n▶ Stage 1 关键词评分...")
        scored = []
        for it in items:
            text = it.get('title', '') + ' ' + it.get('summary', '')
            s = stage1_score(text)
            scored.append((s, it))

        # 按Stage 1分数排序
        scored.sort(key=lambda x: x[0], reverse=True)

        # 取前30条进入Stage 2
        top30 = scored[:30]
        rest = scored[30:]

        print(f"  前30名分数范围: {top30[-1][0]:.0f}-{top30[0][0]:.0f}")
        print(f"  其余 {len(rest)} 条直接设为 reject")

        # 其余全部reject
        for _, it in rest:
            it['yesorno'] = 'reject'

        # Stage 2: DeepSeek 对前30条评分
        print(f"\n▶ Stage 2 DeepSeek 评分（30条）...")
        top_items = [it for _, it in top30]
        for i, it in enumerate(top_items):
            print(f"  [{i+1}] {it.get('title','')[:40]}")

        score_map = call_deepseek(top_items)
        if score_map:
            for local_idx, it in enumerate(top_items):
                score = score_map.get(local_idx + 1, 50)
                if score >= 65:
                    it['yesorno'] = 'strong_select'
                elif score >= 45:
                    it['yesorno'] = 'select'
                elif score >= 25:
                    it['yesorno'] = 'backup'
                else:
                    it['yesorno'] = 'reject'
            print(f"  ✅ DeepSeek 返回成功")
        else:
            print(f"  ⚠️ DeepSeek 无返回，使用 Stage 1 分数")
            for s1, it in top30:
                if s1 >= 65:
                    it['yesorno'] = 'strong_select'
                elif s1 >= 45:
                    it['yesorno'] = 'select'
                elif s1 >= 25:
                    it['yesorno'] = 'backup'
                else:
                    it['yesorno'] = 'reject'

        # 统计
        counts = {'strong_select': 0, 'select': 0, 'backup': 0, 'reject': 0}
        for it in items:
            v = it.get('yesorno', '')
            if v in counts:
                counts[v] += 1
        print(f"\n▶ 最终分布:")
        for k, v in counts.items():
            print(f"  {k}: {v}条")

        break

    with open(path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

    print(f"\n✅ 已更新 {path}")


if __name__ == '__main__':
    main()
