#!/usr/bin/python3.12
"""快速去重：同URL/同标题/标题85%相似度，不用DeepSeek"""
import json, sys, os, re
from datetime import datetime, timedelta
from difflib import SequenceMatcher

input_path = sys.argv[1]
# 默认同时比对 yes_data + learn_records
compare_dirs = sys.argv[2:4] if len(sys.argv) >= 4 else (
    sys.argv[2:3] if len(sys.argv) >= 3 else ['/data/news/json/yes_data/', '/data/news/learn_records/']
)
if not compare_dirs or len(compare_dirs) == 1 and not compare_dirs[0]:
    compare_dirs = ['/data/news/json/yes_data/', '/data/news/learn_records/']

now = datetime.now()

# 加载待去重文件
with open(input_path) as f:
    data = json.load(f)

# 找历史文件（遍历所有比对目录）
recent_files = []
compare_dir = compare_dirs[0]  # keep compatible with downstream logic
for cd in compare_dirs:
    cd = cd.rstrip('/')
    if not os.path.isdir(cd):
        print(f"  ⚠️ 目录不存在: {cd}")
        continue
    for fname in os.listdir(cd):
        if not fname.endswith('.json'):
            continue
        full = os.path.join(cd, fname)
        if full == input_path:
            continue
        m = re.match(r'(\d{4})', fname)
        if not m:
            continue
        mmdd = m.group(1)
        try:
            fdate = datetime.strptime(f"2026-{mmdd[:2]}-{mmdd[2:]}", "%Y-%m-%d")
        except:
            continue
        if (now - fdate) <= timedelta(days=7):
            recent_files.append(full)

# 加载历史items
history = []  # [(title_lower, title_raw, source_url, summary)]
for rf in sorted(recent_files):
    with open(rf) as f:
        his = json.load(f)
    for entry in his if isinstance(his, list) else []:
        for item in entry.get('list', []):
            t = item.get('title', '').strip()
            if t:
                history.append((t.lower(), t, item.get('source_url',''), item.get('summary','')))

print(f"历史数据共 {len(history)} 条")

total_removed = 0
total_kept = 0

for entry in data:
    lst = entry.get('list', [])
    sec_name = entry.get('name', '')
    if not lst:
        print(f"  {sec_name}: 0 -> 0")
        continue
    
    kept = []
    removed = 0
    for item in lst:
        title = item.get('title', '').strip()
        url = item.get('source_url', '')
        summary = item.get('summary', '').strip()
        title_lower = title.lower()
        
        dup = False
        # 同URL去重
        for ht, htr, hu, hs in history:
            if url and hu and url == hu:
                dup = True
                break
        if dup:
            removed += 1
            continue
        
        # 同标题去重
        if title_lower in {h[0] for h in history}:
            dup = True
            removed += 1
            continue
        
        # 标题相似度≥85%
        for ht, htr, hu, hs in history:
            if len(title) >= 3 and len(htr) >= 3:
                r = SequenceMatcher(None, title_lower, ht).ratio()
                if r >= 0.70:
                    dup = True
                    break
        
        if dup:
            removed += 1
        else:
            kept.append(item)
    
    entry['list'] = kept
    total_removed += removed
    total_kept += len(kept)
    print(f"  {sec_name}: {len(lst)} -> {len(kept)} 条 (去重 {removed} 条)")

# 写回
with open(input_path, 'w', encoding='utf-8') as f:
    json.dump(data, f, ensure_ascii=False, indent=2)

print(f"\n✅ 共去重 {total_removed} 条, 保留 {total_kept} 条")
print(f"✅ 已保存: {input_path}")
