"""对 origin_data 的 brand/vehicle/social 板块做 Stage 1+Stage 2 评分
输出: 每条item 的 yesorno 字段设置为 strong_select / select / backup / reject
依据: yes_protocol.md — 评分建议映射
"""

import json, os, re, urllib.request
from datetime import datetime

DEEPSEEK_API_URL = 'https://api.deepseek.com/v1/chat/completions'

# 关键词维度表
KEYWORD_MAP = {
    '体育营销': ['世界杯', '奥运', '体育', '赛事', '运动员', '欧冠', '决赛', '球迷', '观赛', '比赛', '球场'],
    '情感营销': ['情感', '共鸣', '温度', '故事', '怀旧', '情怀', '温情', '暖心', '感动', '治愈'],
    '用户运营': ['用户', '私域', '社群', '会员', '粉丝', '圈层', '忠诚', '车主', '社区', '互动', '打卡'],
    '线下体验': ['体验', '场景', '快闪', '线下', '门店', '沉浸', '试驾', '到店', '工厂', '探访'],
    '本土化': ['本土化', '本土', '中国风', '国潮', '传统', '文化', '非遗', '国货', '中国'],
    '联名合作': ['联名', '跨界', 'IP', '联乘', '合作款', '限定', '定制', '联名款', '携手'],
    'AI/数字化': ['AI', '人工智能', '数字化', '数据', '算力', '算法', '大模型', '智能系统', '芯片', '半导体'],
    '明星代言': ['代言', '明星', 'KOL', '艺人', '大使', '官宣', '明星营销'],
    '内容营销': ['短剧', '内容', '短视频', '直播', '共创', 'UGC', '短片', '视频', '广告', '种草'],
    '新车': ['上市', '发布', '预售', '亮相', '首发', '开启预售', '新车'],
    '电动化': ['电动', '纯电', 'EV', '续航', '充电', '新能源', '纯电动'],
    '混动': ['混动', 'PHEV', '增程', '轻混', '混动版', '插混'],
    '智能化': ['智能', '智驾', '自动驾驶', '座舱', '激光雷达', '乾崑', '天枢'],
    '价格/权益': ['售价', '万元', '万起', '补贴', '优惠', '限时', '福利', '降价', '置换', '金融', '低至'],
    '安全/质量': ['安全', '质量', '召回', '碰撞', '维修', '保养', '售后', '保险'],
    '消费/经济': ['消费', '经济', '就业', '收入', '物价', '房价', '补贴', '政策'],
    '出行/交通': ['出行', '交通', '自驾', '旅游', '通勤', '航空', '高铁', '地铁'],
    '世界杯专项': ['世界杯', '赛事', '观赛', '球迷', '足球', '世界杯赞助', '赛场', '球队', '球星'],
}

WEIGHTS = {
    'brand_hotspots': {
        '体育营销': 0.70, '情感营销': 0.68, '用户运营': 0.73, '线下体验': 0.56,
        '本土化': 0.53, '联名合作': 0.46, 'AI/数字化': 0.39, '明星代言': 0.33,
        '内容营销': 0.21, '价格/权益': 0.15,
    },
    'vehicle_hotspots': {
        '用户运营': 0.71, '线下体验': 0.44, '本土化': 0.59, 'AI/数字化': 0.28,
        '内容营销': 0.15, '新车': 0.99, '电动化': 0.74, '混动': 0.38,
        '智能化': 0.55, '价格/权益': 0.65,
    },
    'social_hotspots': {
        '体育营销': 0.39, '用户运营': 0.10, '线下体验': 0.10, '本土化': 0.10,
        'AI/数字化': 0.20, '智能化': 0.10, '价格/权益': 0.08,
        '消费/经济': 0.30, '出行/交通': 0.25,
    },
}

TOTAL_WEIGHTS = {k: sum(v.values()) for k, v in WEIGHTS.items()}

# 评分建议映射阈值
def score_to_suggestion(score):
    if score >= 65:
        return 'strong_select'
    elif score >= 45:
        return 'select'
    elif score >= 25:
        return 'backup'
    else:
        return 'reject'


def get_deepseek_key():
    cfg_path = '/root/.openclaw/openclaw.json'
    try:
        with open(cfg_path) as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except Exception as e:
        print(f"  读取DeepSeek Key失败: {e}")
        return None


def stage1_keyword_score(text, section_name):
    """Stage 1: 关键词加权粗筛"""
    text_lower = text.lower()
    hit_dims = []
    hit_weight = 0.0
    weights = WEIGHTS.get(section_name, {})
    total = TOTAL_WEIGHTS.get(section_name, 1)

    for dim, keywords in KEYWORD_MAP.items():
        if dim not in weights:
            continue
        for kw in keywords:
            if kw.lower() in text_lower:
                hit_dims.append(dim)
                hit_weight += weights[dim]
                break

    raw_score = (hit_weight / total) * 100 if total > 0 else 0
    dim_bonus = min(len(hit_dims) * 5, 20)
    score = min(100, raw_score + dim_bonus)

    # 世界杯周期加成（6月12日~7月20日）
    if '体育营销' in hit_dims or '世界杯专项' in hit_dims:
        now = datetime.now()
        if (now.month == 6 and now.day >= 12) or (now.month == 7 and now.day <= 20):
            score = min(100, score * 1.4)

    return score, hit_dims


def call_deepseek_batch(items, section_name):
    """将一批item发给DeepSeek进行语义评分"""
    key = get_deepseek_key()
    if not key:
        return None

    lines = []
    for i, it in enumerate(items, 1):
        title = it.get('title', '')
        summary = it.get('summary', '')
        lines.append(f"[{i}] 标题：{title}")
        lines.append(f"    摘要：{summary}")
    items_text = "\n".join(lines)

    guide_map = {
        'brand_hotspots': (
            "评分标准（0-100）：\n"
            "90-100：与汽车行业高度相关的品牌营销案例，有创新营销策略或行业影响力\n"
            "70-89：有价值的品牌营销动作，对现代汽车有参考意义\n"
            "50-69：普通品牌动态，有一定参考价值\n"
            "0-49：与汽车行业无关或价值较低\n"
            "注意：优先选体育IP、用户运营、内容共创、跨界联名等可转化到汽车品牌营销的策略"),
        'vehicle_hotspots': (
            "评分标准（0-100）：\n"
            "90-100：重磅新车上市/预售，或影响行业格局的重大事件\n"
            "70-89：重要车型动态（新车发布、技术突破、价格调整等）\n"
            "50-69：普通车型信息\n"
            "0-49：与车型/汽车行业无关\n"
            "注意：关注新车上市、价格变动、技术突破、竞品动态"),
        'social_hotspots': (
            "评分标准（0-100）：\n"
            "90-100：全网级现象级热点，全民热议话题\n"
            "70-89：行业级热点，高阅读量话题\n"
            "50-69：有一定影响力的社会新闻\n"
            "0-49：低价值或与汽车/消费场景无关的内容\n"
            "注意：优先选与汽车消费、出行、消费趋势、体育赛事相关的话题"),
    }
    scoring_guide = guide_map.get(section_name, guide_map['social_hotspots'])

    prompt = f'''你是一位汽车行业热点分析师。请为以下{len(items)}条{section_name}逐条判定综合价值评分。

{scoring_guide}

请严格按照以下JSON格式输出，只输出JSON数组，不要任何其他文字：
[
  {{"idx": 1, "score": 85}},
  {{"idx": 2, "score": 72}},
  ...
]

待评新闻：
{items_text}'''

    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 2000
    }).encode()

    req = urllib.request.Request(
        DEEPSEEK_API_URL, data=payload,
        headers={
            'Authorization': f'Bearer {key}',
            'Content-Type': 'application/json'
        }
    )
    try:
        resp = urllib.request.urlopen(req, timeout=60)
        result = json.loads(resp.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content'].strip()

        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            reply_json = reply[json_start:json_end]
            scores = json.loads(reply_json)
            score_map = {}
            for s in scores:
                idx = s.get('idx')
                score = s.get('score', 0)
                score_map[idx] = max(0, min(100, int(score)))
            return score_map
        print(f"    ⚠️ 无法解析DeepSeek返回: {reply[:100]}")
        return None
    except Exception as e:
        print(f"    ❌ DeepSeek调用失败: {e}")
        return None


def score_section(name, item_list, batch_size=20):
    """对一个板块的所有item进行评分并设置yesorno"""
    total = len(item_list)
    print(f"\n{'='*60}")
    print(f"板块: {name}（共{total}条）")
    print(f"{'='*60}")

    # Stage 1: 关键词粗筛
    print(f"\n▶ Stage 1 关键词加权评分...")
    stage1_scores = []
    for it in item_list:
        text = it.get('title', '') + ' ' + it.get('summary', '')
        s1, hit_dims = stage1_keyword_score(text, name)
        stage1_scores.append(s1)

    print(f"  平均分: {sum(stage1_scores)/len(stage1_scores):.1f}  范围: {min(stage1_scores):.0f}-{max(stage1_scores):.0f}")

    # Stage 2: DeepSeek 语义精筛（分批）
    print(f"\n▶ Stage 2 DeepSeek语义评分...")
    all_scores = {}
    for start in range(0, total, batch_size):
        batch = item_list[start:start + batch_size]
        batch_num = start // batch_size + 1
        total_batches = (total + batch_size - 1) // batch_size
        print(f"  批次 {batch_num}/{total_batches}（{len(batch)}条）...")
        for i, it in enumerate(batch):
            print(f"    [{start+i+1}] {it.get('title','')[:40]}")

        score_map = call_deepseek_batch(batch, name)
        if score_map:
            for local_idx, score in score_map.items():
                global_idx = start + local_idx - 1
                if 0 <= global_idx < total:
                    all_scores[global_idx] = score
            print(f"    ✅ DeepSeek返回成功")
        else:
            print(f"    ⚠️ DeepSeek无返回，使用Stage 1分数")
            for local_idx in range(len(batch)):
                all_scores[start + local_idx] = stage1_scores[start + local_idx]

    # 最终分数
    final_scores = []
    for idx in range(total):
        final_scores.append(all_scores.get(idx, stage1_scores[idx]))

    # 映射到yesorno
    suggest_counts = {'strong_select': 0, 'select': 0, 'backup': 0, 'reject': 0}
    for idx, item in enumerate(item_list):
        score = final_scores[idx]
        suggestion = score_to_suggestion(score)
        item['yesorno'] = suggestion
        suggest_counts[suggestion] += 1

    print(f"\n▶ 评分结果分布:")
    for k, v in suggest_counts.items():
        print(f"  {k}: {v}条")
    print(f"  最终分数范围: {min(final_scores):.0f}-{max(final_scores):.0f}")
    print(f"  平均分: {sum(final_scores)/len(final_scores):.1f}")


def main():
    path = '/data/news/json/origin_data/0615data.json'
    if not os.path.exists(path):
        print(f"文件不存在: {path}")
        return

    with open(path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    target = ['brand_hotspots', 'vehicle_hotspots', 'social_hotspots']
    for section in data:
        name = section.get('name', '')
        if name in target:
            item_list = section.get('list', [])
            if name == 'social_hotspots':
                # 社会热点先Stage 1粗筛取top 60进DeepSeek
                score_section(name, item_list, batch_size=30)
            else:
                score_section(name, item_list, batch_size=15)

    with open(path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

    print(f"\n{'='*60}")
    print(f"✅ 评分完成，已写入 {path}")


if __name__ == '__main__':
    main()
