"""社会热点分批评分 → 补全yes_0617data.json"""
import json, re, urllib.request
from datetime import datetime

DEEPSEEK_URL = 'https://api.deepseek.com/v1/chat/completions'
INPUT = '/data/news/json/origin_data/yes_0617data.json'

def get_ds_key():
    with open('/root/.openclaw/openclaw.json') as f:
        return json.load(f)['models']['providers']['deepseek']['apiKey']

SOCIAL_PROMPT = """你是现代汽车营销情报分析师。请严格按以下规则对{count}条社会热点评分并选出最有价值的5条。

## 决策规则
1. 优先：与汽车消费、出行、用车场景直接相关的热点
2. 其次：能转化为营销动作的热点（体育赛事、消费趋势、政策变化）
3. 排除：疑问句标题、PR软文、纯娱乐八卦、政治敏感

## 评分映射
≥65→strong_select, 45-64→select(均→yes), 25-44→backup, <25→reject

## 输出格式（严格JSON数组，所有字段必须保留）
[
  {{
    "rank": 1, "idx": 序号, "selection_suggestion": "select",
    "scores": {{"overall_tag_value_score": 0-100}},
    "selection_reason": "15字内理由", "marketing_opportunity": "20字内启示",
    "event_type": "社会热点|世界杯体育营销|技术合作|其他",
    "business_tags": [], "risk_tags": ["无明显风险"], "confidence": 0.85
  }},
  ...
]

待评数据：
{items_text}"""

def call_batch(items):
    key = get_ds_key()
    if not key:
        return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append("[%d] 标题：%s" % (i, it.get('title', '')))
        lines.append("    摘要：%s" % it.get('summary', ''))
    items_text = "\n".join(lines)
    
    prompt = SOCIAL_PROMPT.format(count=len(items), items_text=items_text)
    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3, "max_tokens": 3000
    }).encode()
    
    req = urllib.request.Request(DEEPSEEK_URL, data=payload,
        headers={'Authorization': 'Bearer ' + key, 'Content-Type': 'application/json'})
    resp = urllib.request.urlopen(req, timeout=90)
    reply = json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
    json_start = reply.find('[')
    json_end = reply.rfind(']') + 1
    if json_start >= 0 and json_end > json_start:
        return json.loads(reply[json_start:json_end])
    return None

# ---- Stage 1 关键词评分 ----
KEYWORDS_SOCIAL = {
    '体育营销': ['世界杯','奥运','体育','赛事','运动员','欧冠','决赛','球迷','观赛','比赛','球场'],
    '消费/经济': ['消费','经济','就业','收入','物价','房价','补贴','政策'],
    '出行/交通': ['出行','交通','自驾','旅游','通勤','航空','高铁','地铁'],
    'AI/数字化': ['AI','人工智能','数字化','数据','算力','算法','大模型','智能系统','芯片','半导体'],
    '智能化': ['智能','智驾','自动驾驶','座舱','激光雷达','乾崑','天枢'],
    '价格/权益': ['售价','万元','万起','补贴','优惠','限时','福利','降价','置换','金融','低至'],
}
WEIGHTS = {'体育营销':0.39,'消费/经济':0.30,'出行/交通':0.25,'AI/数字化':0.20,'智能化':0.10,'价格/权益':0.08}
TOTAL = sum(WEIGHTS.values())

def stage1(text):
    text_lower = text.lower()
    hit_w = 0.0; hit_n = 0
    for dim, keywords in KEYWORDS_SOCIAL.items():
        w = WEIGHTS.get(dim, 0)
        for kw in keywords:
            if kw.lower() in text_lower:
                hit_w += w; hit_n += 1; break
    s = min(100, (hit_w / TOTAL) * 100 + min(hit_n * 5, 20))
    now = datetime.now()
    if (now.month == 6 and now.day >= 12) or (now.month == 7 and now.day <= 20):
        for kw in ['世界杯','球迷','足球']:
            if kw.lower() in text_lower:
                s = min(100, s * 1.4); break
    return s

with open(INPUT, 'r', encoding='utf-8') as f:
    data = json.load(f)

for section in data:
    if section.get('name') != 'social_hotspots':
        continue
    items = section.get('list', [])
    print("社会热点共%d条" % len(items))
    
    # Stage 1取前30
    scored = [(stage1(it.get('title','')+' '+it.get('summary','')), it) for it in items]
    scored.sort(key=lambda x: x[0], reverse=True)
    top30 = [it for _, it in scored[:30]]
    
    # 清空yesorno
    for it in items:
        it['yesorno'] = ''
        for f in ['protocol_scores','selection_suggestion','selection_reason','marketing_opportunity','business_tags']:
            it.pop(f, None)
    
    # 分批DeepSeek
    all_results = []
    for start in range(0, 30, 15):
        batch = top30[start:start+15]
        bn = start//15 + 1
        print("  批次%d/2（%d条）..." % (bn, len(batch)))
        r = call_batch(batch)
        if r and len(r) == len(batch):
            all_results.extend(r)
            print("    ✅")
        else:
            print("    ⚠️ 跳过")
            all_results.extend([None]*len(batch))
    
    # 取Top 5
    scored_items = []
    for idx, it in enumerate(top30):
        r = all_results[idx] if idx < len(all_results) else None
        if not r:
            continue
        overall = r.get('scores', {}).get('overall_tag_value_score', 0)
        if overall >= 45:
            it['yesorno'] = 'yes'
            it['protocol_scores'] = r.get('scores', {})
            it['selection_suggestion'] = r.get('selection_suggestion', '')
            it['selection_reason'] = r.get('selection_reason', '')
            it['marketing_opportunity'] = r.get('marketing_opportunity', '')
            it['event_type'] = r.get('event_type', '')
            it['business_tags'] = r.get('business_tags', [])
            it['risk_tags'] = r.get('risk_tags', [])
            it['confidence'] = r.get('confidence', 0.85)
            scored_items.append((overall, it))
    
    scored_items.sort(key=lambda x: x[0], reverse=True)
    # 只保留top 5为yes
    for i, (_, it) in enumerate(scored_items):
        if i >= 5:
            it['yesorno'] = ''
    
    yes_count = sum(1 for it in items if it.get('yesorno') == 'yes')
    print("\n  ✅ 结果: yes=%d" % yes_count)
    for it in items:
        if it.get('yesorno') == 'yes':
            s = it.get('protocol_scores', {})
            print("    overall=%-3s %s" % (s.get('overall_tag_value_score',''), it.get('title','')[:45]))
    break

with open(INPUT, 'w', encoding='utf-8') as f:
    json.dump(data, f, ensure_ascii=False, indent=2)
print("\n✅ 已保存 %s" % INPUT)
