"""合并+去重+完整协议评分 → yes_0617data.json（全字段输出）"""
import json, os, copy, re, urllib.request
from datetime import datetime, timedelta

ORIGIN = '/data/news/json/origin_data/0617data.json'
ALL = '/data/news/json/origin_data/all_0617data.json'
OUTPUT = '/data/news/json/origin_data/yes_0617data.json'
YES_DIR = '/data/news/json/yes_data/'
DEEPSEEK_URL = 'https://api.deepseek.com/v1/chat/completions'

def get_ds_key():
    try:
        with open('/root/.openclaw/openclaw.json') as f:
            return json.load(f)['models']['providers']['deepseek']['apiKey']
    except:
        return None

# ---- 1. 合并 ----
print("=" * 60)
print("第1步：合并两个数据源")
print("=" * 60)

with open(ORIGIN, 'r', encoding='utf-8') as f:
    d1 = json.load(f)
with open(ALL, 'r', encoding='utf-8') as f:
    d2 = json.load(f)

merged = {}
for entry in d1 + d2:
    name = entry.get('name', '')
    items = entry.get('list', [])
    if name not in merged:
        merged[name] = []
    merged[name].extend(items)

for k, v in merged.items():
    print("  %s: %d条" % (k, len(v)))

# ---- 2. 去重（URL + 标题精确） ----
print("\n" + "=" * 60)
print("第2步：去重")
print("=" * 60)

history_items = []
for fn in sorted(os.listdir(YES_DIR)):
    if not fn.endswith('.json') or fn == 'data.json':
        continue
    try:
        with open(os.path.join(YES_DIR, fn), 'r', encoding='utf-8') as f:
            hd = json.load(f)
        if isinstance(hd, list):
            for sec in hd:
                for it in sec.get('list', []):
                    t = (it.get('title') or '').strip()
                    if t:
                        history_items.append({
                            'title': t, 'summary': (it.get('summary') or '').strip(),
                            'brand': (it.get('brand') or '').strip(),
                            'source_url': (it.get('source_url') or '').strip()
                        })
        elif isinstance(hd, dict):
            for lst in hd.values():
                if isinstance(lst, list):
                    for it in lst:
                        if isinstance(it, dict):
                            t = (it.get('title') or '').strip()
                            if t:
                                history_items.append({
                                    'title': t, 'summary': (it.get('summary') or '').strip(),
                                    'brand': (it.get('brand') or '').strip(),
                                    'source_url': (it.get('source_url') or '').strip()
                                })
    except:
        pass
print("  历史数据: %d条" % len(history_items))

hist_titles = {h['title'].lower().strip() for h in history_items}
hist_urls = {h['source_url'] for h in history_items if h['source_url']}

for name in list(merged.keys()):
    items = merged[name]
    seen_urls = set(); seen_titles = set(); kept = []
    for it in items:
        url = (it.get('source_url') or '').strip()
        title = (it.get('title') or '').strip()
        if (url and (url in hist_urls or url in seen_urls)) or \
           (title.lower().strip() in hist_titles or title.lower().strip() in seen_titles):
            continue
        kept.append(it)
        if url: seen_urls.add(url)
        if title: seen_titles.add(title.lower().strip())
    if len(kept) < len(items):
        print("  %s: %d → %d (去重 %d)" % (name, len(items), len(kept), len(items)-len(kept)))
    merged[name] = kept

# ---- 3. 完整协议评分 ----
print("\n" + "=" * 60)
print("第3步：DeepSeek完整协议评分")
print("=" * 60)

PROTOCOL_PROMPT = """你是现代汽车营销情报分析师。请严格按以下规则对{count}条{section}进行评分。

## 决策规则
1. 优先选与汽车消费、用车场景、出行相关的话题
2. 其次选能转化为营销动作的热点（体育赛事、消费趋势、跨界合作）
3. 再次选具有高话题度的社会事件
4. 排除：PR软文、疑问句观点文、纯娱乐八卦、政治敏感

## 车型板块特殊规则
- 有具体车型的营销模式创新（直播带货/跨界主播卖车等）评分最高

## 评分映射
- ≥ 65: strong_select → yes
- 45-64: select → yes
- 25-44: backup
- < 25: reject

## 请按价值从高到低排序，{section}选前5条，其余标记为reject

## 输出格式（严格JSON数组，保留所有字段）
[
  {{
    "rank": 1, "idx": 序号,
    "scores": {{
      "hyundai_relevance_score": 0-100, "conversion_value_score": 0-100,
      "marketing_actionability_score": 0-100, "brand_influence_score": 0-100,
      "customer_loyalty_score": 0-100, "competitor_threat_score": 0-100,
      "overall_tag_value_score": 0-100
    }},
    "selection_suggestion": "",
    "selection_reason": "15字内理由",
    "marketing_opportunity": "20字内营销启示",
    "event_type": "品牌营销案例 | 竞品上市 | 竞品预售 | 竞品降价 | 世界杯体育营销 | 体育营销 | 社会热点 | 技术合作 | 明星代言 | 私域运营 | 其他",
    "industry_tags": [], "business_tags": [], "marketing_tags": [], "target_user_tags": [], "risk_tags": ["无明显风险"],
    "confidence": 0.85
  }},
  ...
]

待评{section}：
{items_text}"""

def call_ds(items, section_name, is_full_list=False):
    key = get_ds_key()
    if not key:
        return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append("[%d] 标题：%s" % (i, it.get('title', '')))
        lines.append("    摘要：%s" % it.get('summary', ''))
        lines.append("    品牌：%s" % it.get('brand', ''))
    items_text = "\n".join(lines)
    
    target_count = "所有" if is_full_list else "前15"
    prompt = PROTOCOL_PROMPT.format(count=len(items), section=section_name, items_text=items_text)
    
    payload = json.dumps({
        "model": "deepseek-chat",
        "messages": [{"role": "user", "content": prompt}],
        "temperature": 0.3,
        "max_tokens": 4000
    }).encode()
    
    req = urllib.request.Request(DEEPSEEK_URL, data=payload,
        headers={'Authorization': 'Bearer ' + key, 'Content-Type': 'application/json'})
    try:
        resp = urllib.request.urlopen(req, timeout=120)
        reply = json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
        json_start = reply.find('[')
        json_end = reply.rfind(']') + 1
        if json_start >= 0 and json_end > json_start:
            return json.loads(reply[json_start:json_end])
        print("  ⚠️ 解析失败: %s" % reply[:200])
        return None
    except Exception as e:
        print("  ❌ %s" % e)
        return None

def has_model(it):
    m = (it.get('model') or '').strip()
    if m: return True
    t = it.get('title', '')
    for p in [r'[A-Z][0-9]', r'[0-9]+款\s*[^\s]+', r'猎手K[0-9]+',
               r'钛7|泰山X8|豪越L|星途EX6|凡尔赛C5|银河战舰|欧拉7|MR2|贝塔']:
        if re.search(p, t): return True
    return False

section_order = ['brand_hotspots', 'vehicle_hotspots', 'social_hotspots',
                 'hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international']
output = []

for name in section_order:
    items = merged.get(name, [])
    sec = {"name": name, "opinion": "", "list": items}
    
    if name not in ('brand_hotspots', 'vehicle_hotspots', 'social_hotspots'):
        output.append(sec)
        continue
    
    print("\n【%s】%d条" % (name, len(items)))
    
    # 车型先过滤有车型的
    if name == 'vehicle_hotspots':
        model_items = [it for it in items if has_model(it)]
        print("  有车型: %d条" % len(model_items))
        scored = [(len(it.get('title','')), it) for it in model_items]  # 简单保序
        target = model_items[:15]
    else:
        target = items[:15] if name == 'brand_hotspots' else items[:30]
    
    # 全部设为空
    for it in items:
        it['yesorno'] = ''
    
    results = call_ds(target, name)
    if not results:
        print("  ❌ DeepSeek失败")
        output.append(sec)
        continue
    
    # 将评分写入items
    for r in results[:5]:  # top 5
        idx = r.get('idx', 1) - 1
        if 0 <= idx < len(target):
            it = target[idx]
            scores = r.get('scores', {})
            it['protocol_scores'] = scores
            it['selection_suggestion'] = r.get('selection_suggestion', '')
            it['selection_reason'] = r.get('selection_reason', '')
            it['marketing_opportunity'] = r.get('marketing_opportunity', '')
            it['event_type'] = r.get('event_type', '')
            it['industry_tags'] = r.get('industry_tags', [])
            it['business_tags'] = r.get('business_tags', [])
            it['marketing_tags'] = r.get('marketing_tags', [])
            it['target_user_tags'] = r.get('target_user_tags', [])
            it['risk_tags'] = r.get('risk_tags', [])
            it['confidence'] = r.get('confidence', 0.85)
            
            overall = scores.get('overall_tag_value_score', 0)
            it['yesorno'] = 'yes' if overall >= 45 else ('backup' if overall >= 25 else '')
    
    yes_n = sum(1 for it in items if it.get('yesorno') == 'yes')
    print("  top5评分完成, yes=%d" % yes_n)
    
    # 打印评分详情
    for it in items:
        if it.get('yesorno') == 'yes':
            s = it.get('protocol_scores', {})
            print("  ✅ overall=%-3s %s" % (s.get('overall_tag_value_score',''), it.get('title','')[:45]))
            print("     H=%-2d C=%-2d M=%-2d B=%-2d L=%-2d T=%-2d  理由:%s" % (
                s.get('hyundai_relevance_score',0), s.get('conversion_value_score',0),
                s.get('marketing_actionability_score',0), s.get('brand_influence_score',0),
                s.get('customer_loyalty_score',0), s.get('competitor_threat_score',0),
                it.get('selection_reason','')))
    
    output.append(sec)

with open(OUTPUT, 'w', encoding='utf-8') as f:
    json.dump(output, f, ensure_ascii=False, indent=2)

print("\n" + "=" * 60)
print("✅ 已保存 %s" % OUTPUT)
print("=" * 60)
