#!/usr/bin/python3.12
"""
社会热点评分：先走本地规则，阈值内拿不准的再走豆包大模型决策。

流程:
  1. 规则过滤（排除类关键词直接淘汰）
  2. 规则打分（科技/商业/产业等加分 + 热度值加成）
  3. 高分直接yes，低分直接淘汰，中间地带送豆包决策

用法:
  python3 score_social_by_doubao.py <json文件路径>
"""

import json, os, sys, re, urllib.request

# 豆包 API 配置
DOUBAO_API_KEY = "ark-563a4210-d804-47fa-a37a-40962192119b-ce6db"
DOUBAO_API_URL = "https://ark.cn-beijing.volces.com/api/v3/chat/completions"
DOUBAO_MODEL = "doubao-seed-2-1-pro-260628"

# ============================================================
#  第一阶段：本地规则过滤 + 打分
# ============================================================

# 绝对排除关键词（标题出现即淘汰）
EXCLUDE_TITLE = [
    '外交部', '国防部', '乌克兰', '利沃夫', '骚乱', '反征兵',
    '美国对中国', '中方回应', '外交部回应', '国防部回应',
    '日本周边', '海军舰艇', '导弹试射', '南海',
    '内马尔', '退役', '体检报告', '哈兰德', '鲁尼', '阿勒代斯',
    '八卦', '绯闻', '综艺', '综艺节目',
    '电蚊拍', '晕倒', '店家回应', '交通事故', '社会治安',
    '自然灾害', '洪涝', '地震', '台风',
    '高考', '中考', '招生', '录取通知书', '中考成绩',
    '企事业单位居家办公', '防汛响应', '防汛抗洪',
    '保时捷', '蔚来', '特斯拉', '理想', '小鹏', '比亚迪',
]

# 来源排除
EXCLUDE_SOURCES = ['news.cn', 'xinhuanet', 'gov.cn', 'people.com.cn', 'cctv.com', 'chinanews', 'china.com']

# 规则加分关键词分类
SCORE_RULES = [
    # (关键词列表, 加分, 分类名)
    (['芯片', '半导体', 'AI', '人工智能', '机器人', '脑机', '量子', '大模型',
      '深度学习', '自动驾驶', '具身智能', '人形机器人', '大语言模型', '算力'], 30, '硬科技'),
    (['AI应用', '数字化', '智能体', '人工智能应用', 'AI工具', '大模型应用',
      'AI赋能', 'AI+', 'AIGC', '智能'], 25, 'AI/数字化'),
    (['出海', '全球化', '出口', '海外市场', '国际贸易', '跨境',
      '产业链', '供应链', '产能', '制造', '工业'], 20, '商业/产业'),
    (['新能源', '光伏', '风电', '储能', '电池', '电动车',
      '新能源汽车', '绿色', '低碳', '碳中和', '清洁能源'], 20, '新能源'),
    (['GDP', '经济', '消费', '零售', 'CPI', 'PMI', '通胀', '就业',
      '金融', '股市', 'A股', '投资', '基金', '人民币'], 15, '宏观经济'),
    (['世界杯', '奥运会', '世锦赛', '赛事', '体育', '电竞',
      '夺冠', '金牌', '冠军'], 5, '体育宏观热度'),
    (['政策', '规划', '方案', '意见', '通知', '措施', '部署',
      '推动', '促进', '支持', '鼓励', '规范'], 10, '政策利好'),
    (['品牌', '联名', '代言', '营销', '广告', 'IP', '跨界',
      '新品', '发布', '首发', '首秀'], 10, '品牌营销'),
]


def rule_score(title):
    """基于规则对社会热点打分，返回 (score, 分类列表)"""
    total = 0
    categories = []
    for keywords, points, cat in SCORE_RULES:
        for kw in keywords:
            if kw in title:
                total += points
                if cat not in categories:
                    categories.append(cat)
                break  # 同一分类只加一次分
    return total, categories


def extract_heat(title):
    """从标题末尾提取热度值（XXX万）"""
    m = re.search(r'([\d.]+)万\s*$', title)
    return float(m.group(1)) if m else 0


# ============================================================
#  第二阶段：豆包决策 prompt
# ============================================================

DOUBAO_PROMPT = """你是一位汽车品牌营销分析师，需要判断以下社会热点是否对营销人有参考价值。

{items}

请为每条输出JSON数组：
[{{"idx":1,"verdict":"yes/backup/no","score":0-100,"reason":"15字内理由"}},...]

规则：
- 偏营销（可借势）：硬科技、AI应用、商业出海、产业政策、宏观经济、可借势话题 → yes(>=70)
- 候选（边缘话题）→ backup(50-69)
- 排除：民生纠纷、娱乐八卦、体育结果、国际政治、纯政策文件、教育新闻、天气灾害 → no(<50)"""


def call_doubao_batch(items_batch):
    """批量调用豆包决策"""
    lines = []
    for i, (it, score, cats, heat) in enumerate(items_batch, 1):
        lines.append(f"[{i}] 标题：{it.get('title','')[:60]}")
    
    prompt = DOUBAO_PROMPT.format(items='\n'.join(lines))
    payload = json.dumps({
        "model": DOUBAO_MODEL,
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 2000
    }).encode('utf-8')

    req = urllib.request.Request(
        DOUBAO_API_URL, data=payload,
        headers={'Authorization': f'Bearer {DOUBAO_API_KEY}', 'Content-Type': 'application/json'}
    )

    try:
        resp = urllib.request.urlopen(req, timeout=300)
        reply = json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
        js = reply.find('[')
        je = reply.rfind(']') + 1
        if js >= 0 and je > js:
            return json.loads(reply[js:je])
        return None
    except Exception as e:
        print(f"    ❌ 豆包调用失败: {e}")
        return None


# ============================================================
#  主流程
# ============================================================

def main():
    import sys
    sys.stdout.reconfigure(line_buffering=True)
    
    if len(sys.argv) < 2:
        print("用法: python3 score_social_by_doubao.py <json文件路径>")
        sys.exit(1)

    json_path = sys.argv[1]
    with open(json_path, 'r', encoding='utf-8') as f:
        d = json.load(f)

    # 提取 social_hotspots
    if isinstance(d, list):
        social_sec = None
        for entry in d:
            if isinstance(entry, dict) and entry.get('name') == 'social_hotspots':
                social_sec = entry
                break
        items = social_sec.get('list', []) if social_sec else []
        is_list_format = True
    elif isinstance(d, dict):
        items = d.get('social_hotspots', [])
        is_list_format = False
    else:
        print("❌ 不支持的JSON格式")
        sys.exit(1)

    print(f"📊 社会热点共 {len(items)} 条")

    # === Phase 1: 规则过滤 + 打分 ===
    results = []  # [(item, rule_score, categories, heat)]
    excluded = 0
    source_excluded = 0

    for it in items:
        title = it.get('title', '')
        url = it.get('source_url', '')

        # 来源排除
        if any(u in url for u in EXCLUDE_SOURCES):
            source_excluded += 1
            continue

        # 标题排除
        excluded_flag = False
        for kw in EXCLUDE_TITLE:
            if kw in title:
                excluded_flag = True
                excluded += 1
                break
        if excluded_flag:
            continue

        score, cats = rule_score(title)
        heat = extract_heat(title)
        # 热度加成：100万以下+1，100万以上每100万+1，最多+10
        heat_bonus = min(int(heat / 100), 10) if heat > 0 else 0
        final_score = score + heat_bonus
        results.append((it, final_score, cats, heat))

    print(f"  来源排除: {source_excluded} | 标题排除: {excluded} | 进入评分: {len(results)}")

    # === Phase 2: 阈值决策 ===
    YES_THRESHOLD = 40     # >=40 直接 yes
    DOUBAO_MIN = 20        # 20~39 送豆包
    # <20 直接淘汰

    auto_yes = []
    doubao_pool = []
    auto_no = []

    for it, score, cats, heat in results:
        if score >= YES_THRESHOLD:
            auto_yes.append((it, score, cats, heat))
        elif score >= DOUBAO_MIN:
            doubao_pool.append((it, score, cats, heat))
        else:
            auto_no.append((it, score, cats, heat))

    print(f"  自动yes: {len(auto_yes)} | 送豆包: {len(doubao_pool)} | 自动排除: {len(auto_no)}")

    # 自动yes按分数排序，取top10
    auto_yes.sort(key=lambda x: x[1], reverse=True)
    for it, score, cats, heat in auto_yes[:10]:
        it['yesorno'] = 'yes'
        it['selection_reason'] = f'规则评分{score}（{",".join(cats)}）'
        it['heat_score'] = 90

    # 送豆包决策（批量送判，最多10条，分2批各5条）
    doubao_pool.sort(key=lambda x: x[1], reverse=True)
    doubao_yes = 0
    doubao_no = 0
    doubao_items = doubao_pool[:10]
    
    for batch_start in range(0, len(doubao_items), 5):
        batch = doubao_items[batch_start:batch_start + 5]
        print(f"  🤖 豆包批量决策（第{batch_start//5+1}批，{len(batch)}条）...")
        doubao_results = call_doubao_batch(batch)
        if doubao_results:
            for r in doubao_results:
                idx = r.get('idx', 0) - 1
                if 0 <= idx < len(batch):
                    it, score, cats, heat = batch[idx]
                    verdict = r.get('verdict', 'no')
                    ds_score = r.get('score', 0)
                    reason = r.get('reason', '')
                    print(f"    [{idx+1}] {verdict}({ds_score}) {it.get('title','')[:35]}")
                    if verdict == 'yes' and ds_score >= 70:
                        it['yesorno'] = 'yes'
                        it['selection_reason'] = f'豆包推荐({ds_score}) {reason}'
                        it['heat_score'] = 90
                        doubao_yes += 1
                    elif verdict == 'backup' or (verdict == 'yes' and ds_score < 70):
                        it['yesorno'] = 'backup'
                        it['selection_reason'] = f'豆包候选({ds_score}) {reason}'
                        it['heat_score'] = 85
                    else:
                        it['yesorno'] = ''
                        it['selection_reason'] = ''
                        doubao_no += 1
        else:
            print(f"    ❌ 本批全部失败")
            for it, _, _, _ in batch:
                it['yesorno'] = ''

    # 热度补充：从剩余未评分条目中按热度补充为 backup
    try:
        scored_titles = {entry.get('title', '') for entry, _, _, _ in results}
    except AttributeError:
        print(f"  ⚠️ results类型异常，调试信息:")
        for i, r in enumerate(results[:5]):
            print(f"    [{i}] type={type(r).__name__}, value={str(r)[:50]}")
        scored_titles = set()
    remaining = [it for it in items if it.get('title', '') not in scored_titles]
    heat_candidates = []
    for it in remaining:
        t = it.get('title', '')
        m = re.search(r'([\d.]+)万\s*$', t)
        if m:
            heat_candidates.append((float(m.group(1)), it))
    heat_candidates.sort(key=lambda x: x[0], reverse=True)
    for heat, it in heat_candidates[:10]:
        it['yesorno'] = 'backup'
        it['selection_reason'] = '热度候选池'

    # 统计
    final_yes = sum(1 for it in items if it.get('yesorno') == 'yes')
    final_backup = sum(1 for it in items if it.get('yesorno') == 'backup')
    print(f"\n✅ 最终: yes={final_yes} backup={final_backup}")
    print(f"   自动yes: {len(auto_yes[:10])} | 豆包新增: {doubao_yes} | 热度候选: {min(10, len(heat_candidates))}")

    print(f"\n=== YES 条目 ===")
    for it in items:
        if it.get('yesorno') == 'yes':
            print(f"  ✅ {it.get('title', '')[:55]}")
            print(f"     理由: {it.get('selection_reason', '')}")

    # 写回
    with open(json_path, 'w', encoding='utf-8') as f:
        json.dump(d, f, ensure_ascii=False, indent=2)
    print(f"\n✅ 已保存: {json_path}")


if __name__ == '__main__':
    main()
