"""修复社会热点评分：重新按正确Stage 1筛选后DeepSeek评分，限定10条
同时修复车型：PR语气过重的降档
"""

import json, re, os, urllib.request
from datetime import datetime

DEEPSEEK_API_URL = 'https://api.deepseek.com/v1/chat/completions'

# 正确Stage 1权重
SOCIAL_WEIGHTS = {
    '体育营销': 0.39, '用户运营': 0.10, '线下体验': 0.10, '本土化': 0.10,
    'AI/数字化': 0.20, '智能化': 0.10, '价格/权益': 0.08, '消费/经济': 0.30, '出行/交通': 0.25,
}
TOTAL_SOCIAL = sum(SOCIAL_WEIGHTS.values())

SOCIAL_KEYWORDS = {
    '体育营销': ['世界杯', '奥运', '体育', '赛事', '运动员', '欧冠', '决赛', '球迷', '观赛', '比赛', '球场'],
    '用户运营': ['用户', '私域', '社群', '会员', '粉丝', '圈层', '忠诚', '车主', '社区', '互动', '打卡'],
    '线下体验': ['体验', '场景', '快闪', '线下', '门店', '沉浸', '试驾', '到店', '工厂', '探访'],
    '本土化': ['本土化', '本土', '中国风', '国潮', '传统', '文化', '非遗', '国货', '中国'],
    'AI/数字化': ['AI', '人工智能', '数字化', '数据', '算力', '算法', '大模型', '智能系统', '芯片', '半导体'],
    '智能化': ['智能', '智驾', '自动驾驶', '座舱', '激光雷达', '乾崑', '天枢'],
    '价格/权益': ['售价', '万元', '万起', '补贴', '优惠', '限时', '福利', '降价', '置换', '金融', '低至'],
    '消费/经济': ['消费', '经济', '就业', '收入', '物价', '房价', '补贴', '政策'],
    '出行/交通': ['出行', '交通', '自驾', '旅游', '通勤', '航空', '高铁', '地铁'],
}


def get_deepseek_key():
    cfg_path = '/root/.openclaw/openclaw.json'
    try:
        with open(cfg_path) as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except:
        return None


def stage1_social_score(text):
    text_lower = text.lower()
    hit_weight = 0.0
    hit_count = 0
    for dim, keywords in SOCIAL_KEYWORDS.items():
        w = SOCIAL_WEIGHTS.get(dim, 0)
        for kw in keywords:
            if kw.lower() in text_lower:
                hit_weight += w
                hit_count += 1
                break
    raw = (hit_weight / TOTAL_SOCIAL) * 100
    bonus = min(hit_count * 5, 20)
    score = min(100, raw + bonus)
    now = datetime.now()
    if (now.month == 6 and now.day >= 12) or (now.month == 7 and now.day <= 20):
        if any(kw in text_lower for kw in ['世界杯', '赛事', '观赛', '球迷', '足球']):
            score = min(100, score * 1.4)
    return score


SOCIAL_PROMPT = """你是现代汽车营销情报分析师。请严格按以下规则对%d条社会热点评分。

## 新增规则（优先级高）
1. ❌ 综合分析文章意图——判断是客观新闻报道还是主观推广PR/软文。考虑整体语气、立场、信息来源，而非止于标题关键词。客观报道（即便含正面描述）可保留，明确PR/软文/通稿倾向的降档。
2. ❌ 减少疑问句/设问标题文章的权重："做对了什么""为什么""如何""怎么""吗？"等提问方式的标题，多为观点文而非事实新闻，自动降档。

## 评分优先级
- 优先：与汽车消费、用车场景、出行直接相关的热点
- 其次：能转化为营销动作的热点（体育赛事、消费趋势、政策变化）
- 再次：具有高话题度的社会事件（可能适合借势）
- 排除：纯娱乐八卦、政治敏感、地方性无关联事件

## 评分映射
- ≥ 65: strong_select → yes
- 45-64: select → yes
- 25-44: backup
- < 25: reject

## 输出格式（严格JSON数组，只输出JSON）
[
  {{
    "idx": 1,
    "selection_suggestion": "select",
    "selection_reason": "15字内理由",
    "marketing_opportunity": "20字内启示",
    "scores": {{"overall_tag_value_score": 0-100}},
    "event_type": "",
    "business_tags": [],
    "risk_tags": ["无明显风险"]
  }},
  ...
]

待评数据：
%s"""


def call_deepseek(items):
    key = get_deepseek_key()
    if not key:
        return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append("[%d] 标题：%s" % (i, it.get('title', '')))
        lines.append("    摘要：%s" % it.get('summary', ''))
    items_text = "\n".join(lines)
    prompt = SOCIAL_PROMPT % (len(items), items_text)
    
    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 4000
    }).encode()
    req = urllib.request.Request(
        DEEPSEEK_API_URL, data=payload,
        headers={'Authorization': 'Bearer ' + key, 'Content-Type': 'application/json'}
    )
    try:
        resp = urllib.request.urlopen(req, timeout=90)
        result = json.loads(resp.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content'].strip()
        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            return json.loads(reply[json_start:json_end])
        print("  ⚠️ 解析失败: %s" % reply[:150])
        return None
    except Exception as e:
        print("  ❌ DeepSeek失败: %s" % e)
        return None


def main():
    path = '/data/news/json/origin_data/0615data.json'
    with open(path, 'r', encoding='utf-8') as f:
        data = json.load(f)
    
    for section in data:
        name = section.get('name', '')
        
        if name == 'social_hotspots':
            items = section.get('list', [])
            print("社会热点共 %d 条" % len(items))
            
            # 正确Stage 1评分
            scored = []
            for it in items:
                text = it.get('title', '') + ' ' + it.get('summary', '')
                s = stage1_social_score(text)
                scored.append((s, it))
            scored.sort(key=lambda x: x[0], reverse=True)
            
            top30 = [it for _, it in scored[:30]]
            print("Stage 1取前30条，范围 %.0f-%.0f" % (scored[29][0], scored[0][0]))
            
            # DeepSeek评分
            print("\n▶ DeepSeek评分（30条）...")
            for i, it in enumerate(top30, 1):
                print("  [%d] %s" % (i, it.get('title','')[:45]))
            
            results = call_deepseek(top30)
            if not results:
                print("DeepSeek失败，跳过")
                continue
            print("  ✅ 返回成功")
            
            # 重置所有social
            for it in items:
                it['yesorno'] = ''
                for f in ['protocol_scores', 'selection_suggestion', 'selection_reason', 'marketing_opportunity']:
                    it.pop(f, None)
            
            # 按overall排序，只取top 10
            scored_by_ds = []
            for idx, it in enumerate(top30):
                r = results[idx] if idx < len(results) else None
                if r and isinstance(r, dict):
                    overall = r.get('scores', {}).get('overall_tag_value_score', 0)
                    scored_by_ds.append((overall, it, r))
            
            scored_by_ds.sort(key=lambda x: x[0], reverse=True)
            top10 = scored_by_ds[:10]
            
            updates = {'yes': 0, 'backup': 0}
            for overall, it, r in top10:
                if overall >= 45:
                    it['yesorno'] = 'yes'
                    updates['yes'] += 1
                elif overall >= 25:
                    it['yesorno'] = 'backup'
                    updates['backup'] += 1
                
                it['protocol_scores'] = r.get('scores', {})
                it['selection_suggestion'] = r.get('selection_suggestion', '')
                it['selection_reason'] = r.get('selection_reason', '')
                it['marketing_opportunity'] = r.get('marketing_opportunity', '')
                it['event_type'] = r.get('event_type', '')
                it['business_tags'] = r.get('business_tags', [])
                it['risk_tags'] = r.get('risk_tags', [])
            
            print("\n▶ 结果：yes=%d backup=%d" % (updates['yes'], updates['backup']))
            for overall, it, r in top10:
                print("    overall=%-3s %s" % (overall, it.get('title','')[:50]))
        
        elif name == 'vehicle_hotspots':
            items = section.get('list', [])
            # 修正PR语气过的条目：30万入手四激光雷达
            for it in items:
                title = it.get('title', '')
                if '四激光雷达' in title and '百万级底盘' in title:
                    if it.get('yesorno') == 'yes':
                        it['yesorno'] = 'backup'
                        print("\n车型降级: %s → backup（PR语气）" % title[:40])
                    break
    
    with open(path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print("\n✅ 已更新 %s" % path)


if __name__ == '__main__':
    main()
