"""重跑社会热点评分 - 分批调用DeepSeek"""
import json, os, urllib.request, copy
from datetime import datetime

DEEPSEEK_API_URL = 'https://api.deepseek.com/v1/chat/completions'
INPUT = '/data/news/json/origin_data/yes_0616data.json'
BATCH_SIZE = 15

def get_ds_key():
    try:
        with open('/root/.openclaw/openclaw.json') as f:
            return json.load(f)['models']['providers']['deepseek']['apiKey']
    except:
        return None

KEYWORDS_SOCIAL = {
    '体育营销': ['世界杯', '奥运', '体育', '赛事', '运动员', '欧冠', '决赛', '球迷', '观赛', '比赛', '球场'],
    '用户运营': ['用户', '私域', '社群', '会员', '粉丝', '圈层', '忠诚', '车主', '社区', '互动', '打卡'],
    '线下体验': ['体验', '场景', '快闪', '线下', '门店', '沉浸', '试驾', '到店', '工厂', '探访'],
    '本土化': ['本土化', '本土', '中国风', '国潮', '传统', '文化', '非遗', '国货', '中国'],
    'AI/数字化': ['AI', '人工智能', '数字化', '数据', '算力', '算法', '大模型', '智能系统', '芯片', '半导体'],
    '智能化': ['智能', '智驾', '自动驾驶', '座舱', '激光雷达', '乾崑', '天枢'],
    '价格/权益': ['售价', '万元', '万起', '补贴', '优惠', '限时', '福利', '降价', '置换', '金融', '低至'],
    '消费/经济': ['消费', '经济', '就业', '收入', '物价', '房价', '补贴', '政策'],
    '出行/交通': ['出行', '交通', '自驾', '旅游', '通勤', '航空', '高铁', '地铁'],
}
WEIGHTS = {'体育营销':0.39,'用户运营':0.10,'线下体验':0.10,'本土化':0.10,'AI/数字化':0.20,'智能化':0.10,'价格/权益':0.08,'消费/经济':0.30,'出行/交通':0.25}
TOTAL = sum(WEIGHTS.values())

def stage1(text):
    text_lower = text.lower()
    hit_w = 0.0; hit_n = 0
    for dim, keywords in KEYWORDS_SOCIAL.items():
        w = WEIGHTS.get(dim, 0)
        for kw in keywords:
            if kw.lower() in text_lower:
                hit_w += w; hit_n += 1; break
    s = min(100, (hit_w / TOTAL) * 100 + min(hit_n * 5, 20))
    now = datetime.now()
    if (now.month == 6 and now.day >= 12) or (now.month == 7 and now.day <= 20):
        for kw in ['世界杯', '球迷', '足球']:
            if kw.lower() in text_lower:
                s = min(100, s * 1.4); break
    return s

def call_batch(items):
    key = get_ds_key()
    if not key:
        return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append("[%d] 标题：%s" % (i, it.get('title', '')))
        lines.append("    摘要：%s" % it.get('summary', ''))
    items_text = "\n".join(lines)
    
    prompt = """你是现代汽车营销情报分析师。请为以下%d条社会热点评分。

评分标准（0-100）：
90-100：与汽车消费/出行/消费趋势高度相关的全网热点
70-89：有影响力的社会话题，可借势营销
50-69：普通社会新闻
0-49：低价值或无关

新增规则：
1. 综合分析文章意图——判断是客观新闻报道还是主观推广PR/软文
2. 减少疑问句/设问标题权重："做对了什么""为什么""如何""怎么""吗？"等提问方式标题自动降档

评分映射：≥65→yes, 45-64→yes, 25-44→backup, <25→reject

输出严格JSON数组：
[{"idx":1,"scores":{"overall_tag_value_score":50},"selection_reason":"15字内","marketing_opportunity":"20字内"},...]

待评数据：
%s""" % (len(items), items_text)

    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 2000
    }).encode()
    
    req = urllib.request.Request(DEEPSEEK_API_URL, data=payload,
        headers={'Authorization': 'Bearer ' + key, 'Content-Type': 'application/json'})
    try:
        resp = urllib.request.urlopen(req, timeout=90)
        reply = json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            return json.loads(reply[json_start:json_end])
        return None
    except Exception as e:
        print("  ❌ %s" % e)
        return None

def main():
    with open(INPUT, 'r', encoding='utf-8') as f:
        data = json.load(f)
    
    for section in data:
        if section.get('name') != 'social_hotspots':
            continue
        items = section.get('list', [])
        print("社会热点共%d条" % len(items))
        
        # Stage 1 取前30
        scored = [(stage1(it.get('title','')+' '+it.get('summary','')), it) for it in items]
        scored.sort(key=lambda x: x[0], reverse=True)
        top30 = [it for _, it in scored[:30]]
        
        # 重置所有yesorno
        for it in items:
            it['yesorno'] = ''
        
        # 分批DeepSeek
        all_results = []
        for start in range(0, 30, BATCH_SIZE):
            batch = top30[start:start+BATCH_SIZE]
            batch_num = start // BATCH_SIZE + 1
            total_batches = (30 + BATCH_SIZE - 1) // BATCH_SIZE
            print("  批次%d/%d（%d条）..." % (batch_num, total_batches, len(batch)))
            r = call_batch(batch)
            if r and len(r) == len(batch):
                all_results.extend(r)
                print("    ✅")
            else:
                print("    ⚠️ 跳过")
                for _ in batch:
                    all_results.append(None)
        
        # 写入评分
        scored_items = []
        for idx, it in enumerate(top30):
            r = all_results[idx] if idx < len(all_results) else None
            if r and isinstance(r, dict):
                overall = r.get('scores', {}).get('overall_tag_value_score', 50)
                if overall >= 45:
                    it['yesorno'] = 'yes'
                elif overall >= 25:
                    it['yesorno'] = 'backup'
                scored_items.append((overall, it))
        
        # 只取top 10为yes
        scored_items.sort(key=lambda x: x[0], reverse=True)
        for idx, (_, it) in enumerate(scored_items):
            it['yesorno'] = 'yes' if idx < 10 else ''
        
        yes_count = sum(1 for it in items if it.get('yesorno') == 'yes')
        print("  yes=%d" % yes_count)
        break
    
    with open(INPUT, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print("✅ 已保存 %s" % INPUT)

if __name__ == '__main__':
    main()
