#!/usr/bin/python3.12
"""与前7天的数据进行去重（标题雷同/一致、summary相似度≥90% 判为重复）
30%以上相似度由 DeepSeek 判断是否为同一事件
用法: python3 remove_duplicates_items.py <待去重文件> <比对目录>
输出: 直接覆写原文件
"""
import json, sys, os, re
from datetime import datetime, timedelta
from difflib import SequenceMatcher
import urllib.request

DEEPSEEK_URL = 'https://api.deepseek.com/v1/chat/completions'


def _get_deepseek_key():
    try:
        with open('/root/.openclaw/openclaw.json') as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except:
        return None


def _ask_deepseek_same_event(t1, t2, s1='', s2='', u1='', u2='', b1='', b2=''):
    """让DeepSeek判断是否为同一事件（含品牌信息辅助判断）"""
    key = _get_deepseek_key()
    if not key:
        return False
    try:
        extra1 = '\n摘要A：' + s1[:100] if s1 else ''
        extra2 = '\n摘要B：' + s2[:100] if s2 else ''
        extra3 = '\n链接A：' + u1 if u1 else ''
        extra4 = '\n链接B：' + u2 if u2 else ''
        extra5 = '\n品牌A：' + b1 if b1 else ''
        extra6 = '\n品牌B：' + b2 if b2 else ''
        prompt = ('判断以下两条新闻是否报道同一事件或同一品牌营销活动。'
                  '如果标题不同但链接相同，则一定是同一事件。'
                  '如果品牌相同或包含相同品牌，即使标题不同也视为同一事件。'
                  '只回复"是"或"否"。'
                  '\n\n标题A：' + t1 + extra1 + extra3 + extra5 +
                  '\n\n标题B：' + t2 + extra2 + extra4 + extra6)
        payload = json.dumps({
            "model": "deepseek-chat",
            "messages": [{"role": "user", "content": prompt}],
            "temperature": 0.1,
            "max_tokens": 10
        }).encode()
        req = urllib.request.Request(DEEPSEEK_URL, data=payload,
            headers={'Authorization': 'Bearer ' + key, 'Content-Type': 'application/json'})
        resp = urllib.request.urlopen(req, timeout=15)
        reply = json.loads(resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
        return '\u662f' in reply  # 是
    except:
        return False


ALL_SECTIONS = {'brand_hotspots', 'vehicle_hotspots', 'social_hotspots',
                'hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international'}
HYUNDAI_SECTIONS = {'hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international'}


def find_recent_files(dir_path, current_path, days=7):
    """在比对目录中找到近7天的文件（排除当前文件自身）"""
    current_base = os.path.basename(current_path)
    now = datetime.now()
    recent_files = []
    if not os.path.isdir(dir_path):
        print(f"⚠️ 比对目录不存在: {dir_path}")
        return recent_files
    for fname in os.listdir(dir_path):
        if not fname.endswith('.json'):
            continue
        if fname == current_base:
            continue
        m = re.match(r'(\d{4})', fname)
        if not m:
            continue
        mmdd = m.group(1)
        try:
            fdate = datetime.strptime(f"2026-{mmdd[:2]}-{mmdd[2:]}", "%Y-%m-%d")
        except:
            continue
        if (now - fdate) <= timedelta(days=days):
            recent_files.append(os.path.join(dir_path, fname))
    return sorted(recent_files)


def load_items(filepath):
    """加载文件中的所有 item（扁平化）"""
    with open(filepath) as f:
        data = json.load(f)
    all_items = []
    if isinstance(data, list):
        for entry in data:
            for item in entry.get('list', []):
                title = (item.get('title') or '').strip()
                summary = (item.get('summary') or '').strip()
                if title:
                    all_items.append({'title': title, 'summary': summary, 'brand': item.get('brand', '') or ''})
    elif isinstance(data, dict):
        for sec_name, lst in data.items():
            if isinstance(lst, list) and lst and isinstance(lst[0], dict) and 'title' in lst[0]:
                for item in lst:
                    title = (item.get('title') or '').strip()
                    summary = (item.get('summary') or '').strip()
                    if title:
                        all_items.append({'title': title, 'summary': summary, 'brand': item.get('brand', '') or ''})
    return all_items


def _normalize_brand(t):
    BRAND_ALIAS = {
        '999感冒灵': '999', '999': '999',
        '耐克': 'NIKE', 'nike': 'NIKE', 'NIKE': 'NIKE',
        '阿迪达斯': 'adidas', 'ADIDAS': 'adidas',
        '腾讯': 'Tencent', '阿里巴巴': 'Alibaba',
        '美团': 'Meituan',
        '小鹏': 'XPeng', 'XPeng': 'XPeng',
    }
    t2 = re.sub(r'\s+', '', t)
    for alias, canonical in sorted(BRAND_ALIAS.items(), key=lambda x: -len(x[0])):
        if alias in t2:
            return t2.replace(alias, canonical), canonical
    return t2, ''


def _clean_core(s):
    return re.sub(r'[\u3010\u3011\[\]\uff5c|\u300a\u300b\u300c\u300d\uff02：:，,。.！!？?——\-…·\s]', '', s)


def is_same_brand_event(t1, t2):
    n1, b1 = _normalize_brand(t1)
    n2, b2 = _normalize_brand(t2)
    if not b1 or not b2 or b1 != b2:
        return False
    core1 = _clean_core(n1.replace(b1, ''))
    core2 = _clean_core(n2.replace(b2, ''))
    short = core1 if len(core1) <= len(core2) else core2
    long_ = core2 if len(core1) <= len(core2) else core1
    if short in long_:
        return True
    if len(short) >= 4:
        return SequenceMatcher(None, short, long_[:len(short)]).ratio() >= 0.55
    return False


def is_duplicate(title, summary, history_titles, history_summaries, section='', history_titles_raw=None, title_summary_map=None, source_url='', history_source_urls=None, brand='', history_brands=None):
    if title_summary_map is None:
        title_summary_map = {}
    """判断是否与历史数据重复（含品牌+URL辅助判重）"""
    title_lower = title.lower().strip()
    
    # 同URL直接判重
    if source_url and history_source_urls and source_url in history_source_urls:
        return True
    
    if title_lower in history_titles:
        return True

    threshold = 0.60 if section in ALL_SECTIONS else 0.75
    ds_threshold = 0.30
    for ht in history_titles:
        ratio = SequenceMatcher(None, title_lower, ht).ratio()
        if ratio >= threshold:
            return True
        if ratio >= ds_threshold and history_titles_raw:
            ht_raw = ''
            for htr in history_titles_raw:
                if htr.lower().strip() == ht:
                    ht_raw = htr
                    break
            if ht_raw:
                hs = title_summary_map.get(ht, '')
                hb = history_brands.get(ht, '') if history_brands else ''
                if _ask_deepseek_same_event(title, ht_raw, summary, hs, source_url, '', brand, hb):
                    return True

    if history_titles_raw:
        for ht in history_titles_raw:
            if is_same_brand_event(title, ht):
                return True

    if summary and history_summaries:
        s_threshold = 0.80 if section in ALL_SECTIONS else 0.90
        for hs in history_summaries:
            if not hs:
                continue
            ratio = SequenceMatcher(None, summary.lower(), hs.lower()).ratio()
            if ratio >= s_threshold:
                return True

    return False


def _clean_title(t):
    t = re.sub(r'[\u3010\u3011\[\]\uff5c|\u300a\u300b\u300c\u300d\uff02\uff1a]', '', t)
    t = re.sub(r'\s+', '', t)
    return t


def dedup_items(items, section=''):
    """同事件去重：hyundai板块保留最详细的一条，30%以上由DeepSeek判定"""
    if section not in HYUNDAI_SECTIONS:
        return items
    
    keep = []
    for item in items:
        t = item.get('title', '')
        tc = _clean_title(t)
        is_dup = False
        for idx, k in enumerate(keep):
            kt = _clean_title(k.get('title', ''))
            short = tc if len(tc) <= len(kt) else kt
            long_ = kt if len(tc) <= len(kt) else tc
            if short in long_:
                is_dup = True
            else:
                r = SequenceMatcher(None, short, long_[:len(short)]).ratio()
                if r >= 0.70:
                    is_dup = True
                elif r >= 0.30:
                    if _ask_deepseek_same_event(t, k.get('title', ''), item.get('summary',''), k.get('summary','')):
                        is_dup = True
            if is_dup:
                if len(t) > len(k.get('title', '')):
                    keep[idx] = item
                break
        if not is_dup:
            keep.append(item)
    return keep


def remove_duplicates(items, history_items, section='', intra_dedup=True):
    """去重（历史去重 + 同文件内去重），返回过滤后的 items 和统计"""
    history_titles = set()
    history_titles_raw = []
    history_summaries = []
    history_source_urls = set()
    history_brands = {}  # title_lower -> brand
    for h in history_items:
        if h['title']:
            history_titles.add(h['title'].lower().strip())
            history_titles_raw.append(h['title'])
            hb = h.get('brand', '') or ''
            if hb:
                history_brands[h['title'].lower().strip()] = hb
        if h['summary']:
            history_summaries.append(h['summary'].lower().strip())
        su = h.get('source_url', '') or ''
        if su:
            history_source_urls.add(su)

    if section in HYUNDAI_SECTIONS:
        items = dedup_items(items, section)

    # 建立 title->summary 映射
    title_summary_map = {}
    for h in history_items:
        ht = (h.get('title') or '').strip().lower()
        hs = h.get('summary', '') or ''
        if ht:
            title_summary_map[ht] = hs

    kept = []
    removed = 0
    for item in items:
        title = (item.get('title') or '').strip()
        summary = (item.get('summary') or '').strip()
        source_url = (item.get('source_url') or '').strip()

        brand = (item.get('brand') or '').strip()

        # 与历史数据去重
        if is_duplicate(title, summary, history_titles, history_summaries, section, history_titles_raw, title_summary_map, source_url, history_source_urls, brand, history_brands):
            removed += 1
            continue

        # 同文件内去重：与已保留的条目比对
        if intra_dedup and kept:
            intra_titles = set()
            intra_titles_raw = []
            intra_summaries = []
            intra_source_urls = set()
            intra_brands = {}
            for k in kept:
                kt = (k.get('title') or '').strip()
                if kt:
                    intra_titles.add(kt.lower().strip())
                    intra_titles_raw.append(kt)
                ks = k.get('summary', '') or ''
                if ks:
                    intra_summaries.append(ks.lower().strip())
                ksu = k.get('source_url', '') or ''
                if ksu:
                    intra_source_urls.add(ksu)
                kb = k.get('brand', '') or ''
                if kb:
                    intra_brands[kt.lower().strip()] = kb
            # 构建同文件 title_summary_map
            intra_map = {}
            for k in kept:
                kt = (k.get('title') or '').strip().lower()
                ks = k.get('summary', '') or ''
                if kt:
                    intra_map[kt] = ks
            if is_duplicate(title, summary, intra_titles, intra_summaries, section, intra_titles_raw, intra_map, source_url, intra_source_urls, brand, intra_brands):
                removed += 1
                continue

        kept.append(item)
    return kept, removed


def main():
    if len(sys.argv) < 2:
        print("用法: python3 remove_duplicates_items.py <待去重文件> [比对目录]")
        sys.exit(1)

    input_path = sys.argv[1]
    compare_dir = sys.argv[2] if len(sys.argv) >= 3 else '/data/news/json/yes_data/'

    if not os.path.exists(input_path):
        print(f"❌ 文件不存在: {input_path}")
        sys.exit(1)

    print(f"[{datetime.now().strftime('%H:%M:%S')}] 加载待去重文件...")
    with open(input_path) as f:
        data = json.load(f)
    print(f"  格式: {type(data).__name__}")

    print(f"  比对目录: {compare_dir}")
    recent_files = find_recent_files(compare_dir, input_path)
    print(f"  找到 {len(recent_files)} 个历史文件")

    history_items = []
    for rf in recent_files:
        try:
            his = load_items(rf)
            history_items.extend(his)
            print(f"    {os.path.basename(rf)}: {len(his)} 条")
        except Exception as e:
            print(f"    ⚠️ {os.path.basename(rf)}: 读取失败 ({e})")

    print(f"\n  历史数据共 {len(history_items)} 条")

    total_removed = 0
    total_kept = 0

    if isinstance(data, list):
        for entry in data:
            lst = entry.get('list', [])
            sec_name = entry.get('name', '')
            kept, removed = remove_duplicates(lst, history_items, sec_name)
            entry['list'] = kept
            total_removed += removed
            total_kept += len(kept)
            print(f"  {sec_name}: {len(lst)} -> {len(kept)} 条 (去重 {removed} 条)")

    with open(input_path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

    print(f"\n✅ 共去重 {total_removed} 条, 保留 {total_kept} 条")
    print(f"✅ 已保存: {input_path}")


if __name__ == '__main__':
    main()
