"""将初筛出的 yes/backup 条目按 yes_protocol.md 完整规范进行 DeepSeek 评分
输出结构化评分数据并写入 items
"""

import json, os, urllib.request
from datetime import datetime

DEEPSEEK_API_URL = 'https://api.deepseek.com/v1/chat/completions'
BATCH_SIZE = 10  # 每批10条，避免token超限


def get_deepseek_key():
    cfg_path = '/root/.openclaw/openclaw.json'
    try:
        with open(cfg_path) as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except:
        return None


PROTOCOL_PROMPT_TEMPLATE = """你是现代汽车营销情报分析师。请严格按以下规则对{count}条{section}进行评分。

## 决策规则
1. 优先选与汽车消费、用车场景、出行相关的话题
2. 其次选能转化为营销动作的热点（体育赛事、消费趋势、政策变化）
3. 再次选具有高话题度的社会事件（可能适合借势）
4. 排除纯娱乐八卦、政治敏感、地方性无关联事件

## 评分映射
- ≥ 65: strong_select（强烈推荐，可直接用于营销）
- 45-64: select（推荐，有价值的情报）
- 25-44: backup（备选，有一定参考价值）
- < 25: reject（不建议使用）

## 标签体系
event_type: 品牌营销案例 | 竞品上市 | 竞品预售 | 竞品降价 | 世界杯体育营销 | 体育营销 | 社会热点 | 技术合作 | 明星代言 | 私域运营 | 其他
industry_tags: 油车 | 轻混 | PHEV | 纯电 | SUV | MPV | 智能驾驶 | 价格 | 经销商 | 售后
business_tags: 提高品牌影响 | 提高销售转化 | 促进试驾 | 促进到店 | 竞品拦截 | 提升用户信任 | 强化家庭用户心智 | 强化低用车成本心智 | 提高客户忠诚度 | 促进留资
marketing_tags: 体育营销 | 内容营销 | 联名营销 | 明星营销 | 小红书种草 | 门店试驾 | 用户共创 | 销售话术 | 热点借势
target_user_tags: 家庭用户 | 年轻用户 | 换购用户 | 价格敏感用户 | 品质型用户 | 老车主
risk_tags: 无明显风险 | 舆情争议 | 政治敏感 | 来源低可信

## 输出格式（严格JSON数组，不要其他文字）
[
  {{
    "idx": 1,
    "selection_suggestion": "strong_select",
    "selection_reason": "简短理由（15字内）",
    "marketing_opportunity": "对现代汽车的营销启示（20字内）",
    "scores": {{
      "hyundai_relevance_score": 0-100,
      "conversion_value_score": 0-100,
      "marketing_actionability_score": 0-100,
      "brand_influence_score": 0-100,
      "customer_loyalty_score": 0-100,
      "competitor_threat_score": 0-100,
      "overall_tag_value_score": 0-100
    }},
    "event_type": "",
    "industry_tags": [],
    "business_tags": [],
    "marketing_tags": [],
    "target_user_tags": [],
    "risk_tags": ["无明显风险"]
  }},
  ...
]

待评{section}数据：
{items_text}"""


def call_deepseek(items, section_name):
    """发送一批items给DeepSeek，按协议规则评分"""
    key = get_deepseek_key()
    if not key:
        return None

    lines = []
    for i, it in enumerate(items, 1):
        title = it.get('title', '')
        summary = it.get('summary', '')
        brand = it.get('brand', it.get('model', ''))
        lines.append(f"[{i}]")
        lines.append(f"    标题：{title}")
        lines.append(f"    摘要：{summary}")
        if brand:
            lines.append(f"    品牌/车型：{brand}")
    items_text = "\n".join(lines)

    prompt = PROTOCOL_PROMPT_TEMPLATE.format(
        count=len(items),
        section=section_name,
        items_text=items_text
    )

    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 4000
    }).encode()

    req = urllib.request.Request(
        DEEPSEEK_API_URL, data=payload,
        headers={
            'Authorization': f'Bearer {key}',
            'Content-Type': 'application/json'
        }
    )
    try:
        resp = urllib.request.urlopen(req, timeout=90)
        result = json.loads(resp.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content'].strip()

        # 提取JSON数组
        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            return json.loads(reply[json_start:json_end])
        print(f"  ⚠️ 无法解析返回，原始内容前200字: {reply[:200]}")
        return None
    except Exception as e:
        print(f"  ❌ DeepSeek调用失败: {e}")
        return None


def score_to_yesorno(overall_score):
    if overall_score >= 65:
        return 'yes'
    elif overall_score >= 45:
        return 'yes'
    elif overall_score >= 25:
        return 'backup'
    else:
        return ''


def main():
    path = '/data/news/json/origin_data/0615data.json'
    with open(path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    target_sections = ['brand_hotspots', 'vehicle_hotspots', 'social_hotspots']

    for section in data:
        name = section.get('name', '')
        if name not in target_sections:
            continue

        items = section.get('list', [])
        # 只处理 yes 或 backup 的条目
        to_score = [it for it in items if it.get('yesorno') in ('yes', 'backup')]
        if not to_score:
            print(f"\n【{name}】无可评分条目")
            continue

        total = len(to_score)
        print(f"\n{'='*60}")
        print(f"【{name}】共{total}条进入DeepSeek协议评分")
        print(f"{'='*60}")

        # 分批调用
        all_results = []
        for start in range(0, total, BATCH_SIZE):
            batch = to_score[start:start + BATCH_SIZE]
            batch_num = start // BATCH_SIZE + 1
            total_batches = (total + BATCH_SIZE - 1) // BATCH_SIZE
            print(f"\n▶ 批次 {batch_num}/{total_batches}（{len(batch)}条）...")
            for i, it in enumerate(batch):
                print(f"  [{start+i+1}] {it.get('title','')[:45]}")

            results = call_deepseek(batch, name)
            if results and len(results) == len(batch):
                all_results.extend(results)
                print(f"  ✅ 返回成功")
            else:
                print(f"  ⚠️ DeepSeek返回异常，跳过本批次")
                # fallback: 保持原 yesorno 不变
                for it in batch:
                    all_results.append(None)

        # 将评分结果写回 items
        updates = {'yes': 0, 'backup': 0, 'reject_to_empty': 0, 'no_change': 0}
        for idx, item in enumerate(to_score):
            result = all_results[idx] if idx < len(all_results) else None
            if result and isinstance(result, dict):
                scores = result.get('scores', {})
                overall = scores.get('overall_tag_value_score', 
                           scores.get('overall_tag_value_score', 50))

                # 更新 yesorno
                old_yesorno = item.get('yesorno', '')
                new_yesorno = score_to_yesorno(overall)
                item['yesorno'] = new_yesorno

                # 写入结构化评分数据
                item['protocol_scores'] = scores
                item['selection_suggestion'] = result.get('selection_suggestion', '')
                item['selection_reason'] = result.get('selection_reason', '')
                item['marketing_opportunity'] = result.get('marketing_opportunity', '')
                item['event_type'] = result.get('event_type', '')
                item['industry_tags'] = result.get('industry_tags', [])
                item['business_tags'] = result.get('business_tags', [])
                item['marketing_tags'] = result.get('marketing_tags', [])
                item['target_user_tags'] = result.get('target_user_tags', [])
                item['risk_tags'] = result.get('risk_tags', [])
                item['selection_reject_reason'] = result.get('reject_reason', '')

                # 统计
                if new_yesorno == 'yes': updates['yes'] += 1
                elif new_yesorno == 'backup': updates['backup'] += 1
                else: updates['reject_to_empty'] += 1
            else:
                updates['no_change'] += 1

        print(f"\n▶ 评分结果：")
        print(f"  yes（选中）: {updates['yes']}条")
        print(f"  backup: {updates['backup']}条")
        print(f"  reject变空: {updates['reject_to_empty']}条")
        if updates['no_change']:
            print(f"  未变更: {updates['no_change']}条")

    # 写入文件
    with open(path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

    # 最终统计
    print(f"\n{'='*60}")
    print("最终各板块 yesorno 分布：")
    for section in data:
        name = section.get('name', '')
        if name in target_sections:
            items = section.get('list', [])
            y = sum(1 for it in items if it.get('yesorno') == 'yes')
            b = sum(1 for it in items if it.get('yesorno') == 'backup')
            e = sum(1 for it in items if it.get('yesorno') == '')
            print(f"  {name}: yes={y}  backup={b}  空={e}")

    print(f"\n✅ 已完成写入 {path}")


if __name__ == '__main__':
    main()
