import json, os, re, urllib.request, copy
from datetime import datetime

DEEPSEEK_URL = 'https://api.deepseek.com/v1/chat/completions'
INPUT_DIR = "/data/news/json/origin_data/"
mmdd2 = datetime.now().strftime("%m%d")
INPUT = f'{INPUT_DIR}/{mmdd2}data.json'
OUTPUT_DIR = "/data/news/json/origin_data/"
mmdd = datetime.now().strftime("%m%d")
OUTPUT = f'{OUTPUT_DIR}/yes_{mmdd}data.json'
YES_DIR = '/data/news/json/yes_data/'

def get_ds_key():
    try:
        with open('/root/.openclaw/openclaw.json') as f:
            return json.load(f)['models']['providers']['deepseek']['apiKey']
    except:
        return None

PROMPT = """为以下{count}条{section}评分（0-100）。排除PR软文和疑问句观点文。

{extra_rules}

输出JSON数组：
[{{"idx":1,"scores":{{"hyundai_relevance_score":0,"conversion_value_score":0,"marketing_actionability_score":0,"brand_influence_score":0,"customer_loyalty_score":0,"competitor_threat_score":0,"overall_tag_value_score":0}},"selection_suggestion":"","selection_reason":"15字内","marketing_opportunity":"20字内","event_type":"","business_tags":[],"risk_tags":["无明显风险"],"confidence":0.85}},...]

{item_text}"""

BRAND_EXTRA = "世界杯赛事期间（6月-7月），与世界杯/体育赛事相关的品牌营销案例可适当加分。重点看品牌营销案例的创意和话题度，不用和汽车沾边。排除汽车品牌案例（归车型板块）。PR风格可适当降分但不直接排除。优先保留有话题性、传播力或创新性的营销活动。"
VEHICLE_PROMPT = """为以下{count}条vehicle_hotspots评分（0-100）。

{extra_rules}

只输出JSON数组，每条格式：{{"idx":序号,"scores":{{"overall_tag_value_score":分数}},"selection_reason":"理由"}}

{item_text}"""
SOCIAL_EXTRA = "偏营销的话题优先，不分行业（只要是可借势做营销的话题优先），其次选高热度社会化热议话题。非品牌化汽车社会新闻（油价、政策调整等）最多1条。排除具体汽车品牌新闻（保时捷/蔚来/特斯拉等归车型板块）、与中国无关的国际政治事件、纯娱乐八卦（知名人士去世除外）。同类事件去重。"

def call_ds(items, section, extra=''):
    key = get_ds_key()
    if not key: return None
    lines = []
    for i, it in enumerate(items, 1):
        lines.append("[%d] 标题：%s" % (i, it.get('title','')))
        lines.append("    品牌：%s" % it.get('brand',''))
    prompt = PROMPT.format(count=len(items), section=section, extra_rules=extra, item_text='\n'.join(lines))
    payload = json.dumps({"model":"deepseek-chat","messages":[{"role":"user","content":prompt}],"temperature":0.3,"max_tokens":3000}).encode()
    req = urllib.request.Request(DEEPSEEK_URL, data=payload, headers={'Authorization':'Bearer '+key,'Content-Type':'application/json'})
    try:
        resp = urllib.request.urlopen(req, timeout=90)
        reply = json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
        js = reply.find('['); je = reply.rfind(']')+1
        if js>=0 and je>js: return json.loads(reply[js:je])
        print("  ⚠️ 解析失败"); return None
    except Exception as e:
        print("  ❌ %s" % e); return None

def has_model(it):
    m = (it.get('model') or '').strip()
    if m: return True
    t = it.get('title','')
    for p in [r'[A-Z][0-9]',r'[0-9]+款',r'猎手K[0-9]+',r'钛7|泰山X8|豪越L|星途|凡尔赛|银河|i-HEV|EV5|i20|NV200|M817|H10|GX7|L03|i3|NX8|Avante|GT7']:
        if re.search(p, t): return True
    return False

with open(INPUT, 'r', encoding='utf-8') as f:
    data = json.load(f)

output = []
for name in ['brand_hotspots','vehicle_hotspots','social_hotspots','hyundai_buzz_topics_domestic','hyundai_buzz_topics_international']:
    items = copy.deepcopy([it for sec in data if sec.get('name')==name for it in sec.get('list',[])])
    if not items:
        output.append({"name":name,"opinion":"","list":[]})
        continue
    print("\n【%s】%d条" % (name, len(items)))
    
    if name == 'brand_hotspots':
        # 时效性过滤：工作日24h，周末48h
        from datetime import datetime as _dt, timedelta as _td
        _now = _dt.now()
        _is_weekend = _now.weekday() >= 5  # 周六5, 周日6
        _max_hours = 48 if _is_weekend else 24
        _cutoff = _now - _td(hours=_max_hours)
        _cutoff_str = _cutoff.strftime('%Y-%m-%d')
        
        # 按日期过滤
        _recent = [it for it in items if (it.get('publish_time','') or '')[:10] >= _cutoff_str]
        _within48 = [it for it in items if (it.get('publish_time','') or '')[:10] >= (_now - _td(hours=48)).strftime('%Y-%m-%d')]
        
        if len(_recent) >= 5:
            top = _recent[:15]
            _time_desc = f'{_max_hours}h内{len(_recent)}条'
        elif len(_within48) >= 5:
            top = _within48[:15]
            _time_desc = f'24h内仅{len(_recent)}条不足，回退至48h内{len(_within48)}条'
        else:
            top = items[:15]
            _time_desc = f'48h内仅{len(_within48)}条不足，使用全部{len(items)}条'
        
        print(f"  📅 {_time_desc}")
        results = call_ds(top, name, BRAND_EXTRA)
        if results:
            for idx, it in enumerate(top):
                r = results[idx] if idx<len(results) else None
                if not r: continue
                o = r.get('scores',{}).get('overall_tag_value_score',50)
                it['yesorno'] = 'yes' if o>=45 else ('backup' if o>=25 else '')
                it['protocol_scores'] = r.get('scores',{})
                it['selection_reason'] = r.get('selection_reason','')
                it['event_type'] = r.get('event_type','')
                it['business_tags'] = r.get('business_tags',[])
    
    elif name == 'vehicle_hotspots':
        with_m = [it for it in items if has_model(it)]
        for it in items: it['yesorno'] = ''
        # 按来源分流，各取最符合的
        autohome_items = [it for it in with_m if 'autohome' in (it.get('source_url','') or '')]
        yiche_items = [it for it in with_m if any(d in (it.get('source_url','') or '') for d in ['yiche', 'bitauto'])]
        # 如果来源划分不全，剩余归入主源
        other_items = [it for it in with_m if it not in autohome_items and it not in yiche_items]
        print(f"    来源: autohome={len(autohome_items)}, yiche={len(yiche_items)}, other={len(other_items)}")
        # 各取前8条（保证双方都有代表）
        top_ah = autohome_items[:8]
        top_yc = yiche_items[:8]
        top = (top_ah + top_yc + other_items)[:15]
        print(f"    采样: autohome {len(top_ah)}条 + yiche {len(top_yc)}条 + other = {len(top)}条")
        # Use simplified prompt for vehicle (7-dim format confuses DeepSeek)
        lines = []
        for i, it in enumerate(top, 1):
            lines.append("[%d] 标题：%s" % (i, it.get('title','')))
        v_prompt = VEHICLE_PROMPT.format(count=len(top), extra_rules="优先选热门新车上市/预售/价格信息及营销模式创新内容，排除观点文和软文。注意H10、F700、ID.7等均为有效车型名。", item_text='\n'.join(lines))
        key = get_ds_key()
        v_payload = json.dumps({"model":"deepseek-chat","messages":[{"role":"user","content":v_prompt}],"temperature":0.3,"max_tokens":2000}).encode()
        v_req = urllib.request.Request(DEEPSEEK_URL, data=v_payload, headers={'Authorization':'Bearer '+key,'Content-Type':'application/json'})
        try:
            v_resp = urllib.request.urlopen(v_req, timeout=90)
            v_reply = json.loads(v_resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
            v_js = v_reply.find('['); v_je = v_reply.rfind(']')+1
            results = json.loads(v_reply[v_js:v_je]) if v_js>=0 and v_je>v_js else None
        except:
            results = None
        if results:
            for idx, it in enumerate(top):
                r = results[idx] if idx<len(results) else None
                if not r: continue
                o = r.get('scores',{}).get('overall_tag_value_score',50)
                it['yesorno'] = 'yes' if o>=45 else ('backup' if o>=25 else '')
                it['protocol_scores'] = r.get('scores',{})
                it['selection_reason'] = r.get('selection_reason','')
                it['event_type'] = r.get('event_type','')
                it['business_tags'] = r.get('business_tags',[])
            # 车型同品牌去重：用DeepSeek语义判断，同品牌/同事件只保留评分最高的
            yes_items = [(it.get('protocol_scores',{}).get('overall_tag_value_score',0), it.get('title',''), it)
                         for it in items if it.get('yesorno') == 'yes']
            if len(yes_items) > 1:
                dedup_lines = []
                for i, (sc, t, it) in enumerate(yes_items, 1):
                    dedup_lines.append(f'[{i}] score={sc} | {t}')
                dedup_prompt = f'''以下车型热点请按"同一品牌只保留一条（不分车型），同一事件只保留一条"去重。品牌相同的无论什么车型都只保留score最高的那条。
输出格式：保留的序号列表 例如 [1,3,5]

{chr(10).join(dedup_lines)}'''
                try:
                    _key = get_ds_key()
                    _payload = json.dumps({"model":"deepseek-chat","messages":[{"role":"user","content":dedup_prompt}],"temperature":0.3,"max_tokens":200}).encode()
                    _req = urllib.request.Request(DEEPSEEK_URL, data=_payload, headers={'Authorization':'Bearer '+_key,'Content-Type':'application/json'})
                    _resp = urllib.request.urlopen(_req, timeout=30)
                    _reply = json.loads(_resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
                    import re as _re2
                    _kept = set()
                    for _m in _re2.findall(r'\d+', _reply):
                        _idx = int(_m) - 1
                        if 0 <= _idx < len(yes_items):
                            _kept.add(_idx)
                    for _i, (_sc, _t, _it) in enumerate(yes_items):
                        if _i not in _kept:
                            _it['yesorno'] = ''
                    print(f"    DeepSeek去重: {len(yes_items)}条 -> {len(_kept)}条")
                except:
                    pass
    
    elif name == 'social_hotspots':
        for it in items: it['yesorno'] = ''
        # 过滤掉官媒/政治类来源
        filtered = [it for it in items if not any(d in (it.get('source_url','') or '') for d in ['news.cn', 'xinhuanet', 'gov.cn', 'people.com.cn', 'cctv.com', 'chinanews', 'china.com'])]
        top45 = filtered[:45] if len(filtered) >= 20 else items[:45]
        results = []
        for s in range(0, len(top45), 15):
            r = call_ds(top45[s:s+15], name, SOCIAL_EXTRA)
            if r: results.extend(r)
        if results:
            scored = []
            # DeepSeek 理由中明确表示排除的，强制降分（弥补 DeepSeek 评分自相矛盾的问题）
            _EXCLUDE_REASONS = ['政治', '外交', '军事', '娱乐八卦', '八卦', '奇闻',
                               '观点文', 'PR软文', '民生', '纠纷', '自然灾害', '教育',
                               '无营销', '无价值']
            for idx, it in enumerate(top45):
                r = results[idx] if idx<len(results) else None
                if not r: continue
                o = r.get('scores',{}).get('overall_tag_value_score',50)
                reason = (r.get('selection_reason','') or '')
                # 理由含排除关键词 → 强制降为 backup
                if any(e in reason for e in _EXCLUDE_REASONS):
                    o = min(o, 25)
                if o>=45: it['yesorno']='yes'
                elif o>=25: it['yesorno']='backup'
                it['selection_reason'] = reason
                it['marketing_opportunity'] = r.get('marketing_opportunity','')
                it['event_type'] = r.get('event_type','')
                it['business_tags'] = r.get('business_tags',[])
                it['risk_tags'] = r.get('risk_tags',[])
                it['confidence'] = r.get('confidence',0.85)
                scored.append((o,it))
            scored.sort(key=lambda x:x[0],reverse=True)
            # 阈值制：≥45→yes, 25~44→backup，不硬凑top10
            _yes_count = 0
            _bak_count = 0
            for o, it in scored:
                if o >= 45:
                    it['yesorno'] = 'yes'
                    _yes_count += 1
                elif o >= 25:
                    it['yesorno'] = 'backup'
                    _bak_count += 1
                else:
                    it['yesorno'] = ''
            # 不足5条时从backup中补足
            if _yes_count < 5:
                _need = 5 - _yes_count
                for o, it in scored:
                    if _need <= 0: break
                    if it.get('yesorno') == 'backup':
                        it['yesorno'] = 'yes'
                        _yes_count += 1
                        _need -= 1
    
    yn = sum(1 for it in items if it.get('yesorno')=='yes')
    bn = sum(1 for it in items if it.get('yesorno')=='backup')
    print("  yes=%d backup=%d" % (yn, bn))
    output.append({"name":name,"opinion":"","list":items})

# URL有效性校验：检查所有yes条目的链接是否可访问，不可访问的取消yes标记
# 验证优先级：curl直连 → r.jina.ai → Playwright
import urllib.request as _url_req, subprocess as _subp, os as _os
print("\n🔍 验证yes条目的链接有效性...")
total_checked = 0
total_invalid = 0

def _verify_url_cmd(url, timeout=8):
    """直接用curl验证链接是否可访问（返回 (True, 内容) 或 (False, 错误)）"""
    try:
        r = _subp.run(['curl', '-s', '-L', '--max-time', str(timeout), url],
                      capture_output=True, timeout=timeout+5)
        if r.returncode != 0:
            return False, f"curl返回码{r.returncode}"
        text = r.stdout.decode('utf-8', errors='ignore')
        if len(text.strip()) < 500:
            return False, f"内容过少({len(text.strip())}字节)"
        body = text[:2000].lower()
        if any(p in body for p in ['<title>404</title>', '<title>页面不存在</title>', '找不到页面', 'error-container', '<h1>404']):
            return False, "页面含错误标记"
        return True, text
    except Exception as e:
        return False, str(e)

def _verify_url_jina(url, timeout=10):
    """用r.jina.ai代理验证"""
    try:
        _req = _url_req.Request(f"https://r.jina.ai/{url}", headers={'User-Agent': 'Mozilla/5.0'})
        _resp = _url_req.urlopen(_req, timeout=timeout)
        _text = _resp.read().decode('utf-8', errors='ignore')
        _body = _text[:2000].lower()
        if len(_text.strip()) < 500 or any(p in _body for p in ['<title>404</title>', '<title>页面不存在</title>', '找不到页面', 'error-container', '<h1>404', '无法访问']):
            return False, f"内容异常({len(_text.strip())}字节)"
        return True, _text
    except Exception as e:
        return False, str(e)

def _verify_url_playwright(url, timeout=15):
    """用Playwright+Stealth验证（针对jina超时/拦截的链接）"""
    try:
        _scr = f'''
import asyncio, sys
sys.path.insert(0, "/root/.local/share/pip/python3.12/site-packages")
from playwright.async_api import async_playwright
import nest_asyncio
nest_asyncio.apply()
async def check():
    async with async_playwright() as p:
        browser = await p.chromium.launch({{"headless": True, "args": ["--no-sandbox", "--disable-setuid-sandbox"]}})
        page = await browser.new_page(viewport={{"width": 1280, "height": 720}}, user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36")
        try:
            await page.goto("{url}", timeout={timeout*1000}, wait_until="domcontentloaded")
            text = await page.inner_text("body")
            text = (text or "")[:2000]
            ok = len(text.strip()) >= 500 and not any(p in text.lower() for p in ['<title>404</title>', '<title>页面不存在</title>', '找不到页面', 'error-container', '<h1>404'])
            await browser.close()
            print("PWOK:" + str(ok))
            if ok: print("PWDATA:" + text)
        except Exception as e:
            await browser.close()
            print("PWERR:" + str(e))
asyncio.run(check())
'''
        r = _subp.run(['/usr/bin/python3.12', '-c', _scr], capture_output=True, timeout=timeout+10)
        out = (r.stdout or b'').decode('utf-8', errors='ignore')
        if 'PWOK:True' in out:
            return True, 'playwright验证通过'
        return False, out.split('PWERR:')[-1][:50] if 'PWERR:' in out else 'playwright验证失败'
    except Exception as e:
        return False, str(e)

for sec in output:
    sec_name = sec.get('name', '')
    is_vehicle = sec_name == 'vehicle_hotspots'
    for li in sec.get('list', []):
        if li.get('yesorno') == 'yes':
            url = li.get('source_url', '')
            if url:
                total_checked += 1
                _text = ''
                _valid = False
                
                # Priority 1: curl直连
                _ok, _result = _verify_url_cmd(url)
                if _ok:
                    _valid = True
                    _text = _result
                
                # Priority 2: jina.ai代理
                if not _valid:
                    _ok, _result = _verify_url_jina(url)
                    if _ok:
                        _valid = True
                        _text = _result
                
                # Priority 3: Playwright（仅jina超时/拦截时, 即上两步都失败）
                if not _valid:
                    _ok, _result = _verify_url_playwright(url)
                    if _ok:
                        _valid = True
                        print(f"  🎭 Playwright验证通过: {li.get('title','')[:35]}")
                
                if not _valid:
                    li['yesorno'] = ''
                    total_invalid += 1
                    print(f"  ❌ 链接无法访问: {li.get('title','')[:40]} ({_result[:30]})")
                    continue
                
                # 车型热点：只看链接有效性，不做时间时效性验证
                if is_vehicle:
                    print(f"  ✅ 链接有效: {li.get('title','')[:35]}")
                    continue
                
                # 非车型：页面有效，提取真实发布时间做时效性验证
                import re as _re
                _real_date = '-'
                _head = _text[:5000]
                for _dt_p in [r'article:published_time["\s>]+content="([^"]+)"',
                               r'<meta[^>]+property="article:published_time"[^>]+content="([^"]+)"',
                               r'pubdate[=:]\s*["\']?(\d{4}-\d{2}-\d{2})',
                               r'datetime="(\d{4}-\d{2}-\d{2})', r'data-time="(\d{4}-\d{2}-\d{2})',
                               r'"datePublished"\s*:\s*"(\d{4}-\d{2}-\d{2})', r'"pubDate"\s*:\s*"(\d{4}-\d{2}-\d{2})',
                               r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})',
                               r'(\d{4}-\d{2}-\d{2})\s*[·]', r'(\d{4}-\d{2}-\d{2})[T ]\d{2}:\d{2}']:
                    _m = _re.search(_dt_p, _head)
                    if _m:
                        _real_date = _m.group(1)[:10]
                        break
                
                if _real_date != '-':
                    from datetime import datetime as _dt, timedelta as _td
                    _cutoff = (_dt.now() - _td(hours=48)).strftime('%Y-%m-%d')
                    if _real_date < _cutoff:
                        li['yesorno'] = ''
                        total_invalid += 1
                        print(f"  ❌ 发布时间({_real_date})超48h: {li.get('title','')[:40]}")
                    else:
                        print(f"  ✅ 发布时间{_real_date}: {li.get('title','')[:35]}")
                else:
                    li['publish_time_verified'] = '-'
                    print(f"  ⚠️ 未获取到发布时间: {li.get('title','')[:35]}")

if total_invalid > 0:
    print(f"  ✅ 共检查{total_checked}条，取消{total_invalid}条无效链接的yes标记")
else:
    print(f"  ✅ 共检查{total_checked}条，全部有效")

# 社会热点：额外按标题热度值标记top10候选池（标记为backup，供用户手动标注）
for sec in output:
    if sec.get('name') == 'social_hotspots':
        candidates = []
        for li in sec.get('list', []):
            if li.get('yesorno') in ('yes', 'backup'):
                continue
            t = li.get('title', '')
            # 排除明显不符合偏营销规则的类别（国际政治、民生纠纷、体育八卦等）
            excludes = ['外交部', '国防部', '乌克兰', '利沃夫', '骚乱', '反征兵',
                       '内马尔', '退役', '电蚊拍', '晕倒', '体检报告']
            if any(e in t for e in excludes):
                continue
            m = __import__('re').search(r'([\d.]+)万\s*$', t)
            if m:
                heat = float(m.group(1))
                candidates.append((heat, li))
        candidates.sort(key=lambda x: x[0], reverse=True)
        for heat, li in candidates[:10]:
            li['yesorno'] = 'backup'
            li['selection_reason'] = '热度候选池'
        print(f"  🔥 社会热点额外标记{min(10, len(candidates))}条热度top10")
        break

with open(OUTPUT,'w',encoding='utf-8') as f:
    json.dump(output,f,ensure_ascii=False,indent=2)
print("\n✅ 已保存 %s" % OUTPUT)
