#!/usr/bin/python3.12
"""从tophub抓取社会热点，写入origin_data social_hotspots"""
import json, os, sys, re, urllib.request
from datetime import datetime, timedelta

ORIGIN_DIR = "/data/news/json/origin_data/"
TOP_URL = "https://r.jina.ai/https://tophub.today/t/m4ejBJMyex"

def fetch_page(page):
    """抓取单页，返回标题+链接列表"""
    url = f"{TOP_URL}?p={page}"
    req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        text = resp.read().decode('utf-8')
    except Exception as e:
        print(f"  ⚠️ p={page} 失败: {e}")
        return []

    articles = []
    seen = set()
    skip_words = ['Image', '登录', '关于我们', 'App下载', '指南', '赞助', 'API', '首页', '日报', '动态',
                  '追踪', '榜中榜', '热文库', '话题', '日历', '综合', '科技', '娱乐', '社区', '购物',
                  '财经', '开发', '简报', '更多', '充值', '成为', '夜间', '过滤', '自动', '管理',
                  '订阅', '创建', '小部件', '自定义', '报刊', '设计', '校务', '政务', '专栏', '苹果',
                  '公众号', '追踪机器人', '我的订阅', '热点', '兴趣', '事件']

    for line in text.split('\n'):
        for m in re.finditer(r'\[([^\]]+)\]\(([^)]+)\)', line):
            t, link = m.group(1).strip(), m.group(2)
            if len(t) < 8 or any(w in t for w in skip_words):
                continue
            if link.startswith('javascript:') or link == '#' or 'tophub.today' in link:
                continue
            if t not in seen:
                seen.add(t)
                articles.append({"title": t, "url": link})

    print(f"  p={page}: 提取 {len(articles)} 条")
    return articles

def get_domain(url):
    """从URL提取平台名"""
    m = re.search(r'https?://([^/]+)', url)
    if m:
        domain = m.group(1)
        # 常见平台映射
        mapping = {
            'zhihu.com': '知乎',
            'so.toutiao.com': '今日头条',
            'bjnews.com.cn': '新京报',
            'news.cctv.com': '央视网',
            'thepaper.cn': '澎湃新闻',
            's.weibo.com': '微博',
            'baijiahao.baidu.com': '百家号',
            'mp.weixin.qq.com': '微信公众号',
            '36kr.com': '36氪',
            'huxiu.com': '虎嗅',
            't.cj.sina.com.cn': '新浪财经',
            'finance.sina.com.cn': '新浪财经',
            'stock.10jqka.com.cn': '同花顺',
            'wallstreetcn.com': '华尔街见闻',
            'jiemian.com': '界面新闻',
            'yicai.com': '第一财经',
            '163.com': '网易',
            'sohu.com': '搜狐',
            'ifeng.com': '凤凰网',
            'guancha.cn': '观察者网',
            'xinwen.bjd.com.cn': '北京日报',
            'people.com.cn': '人民网',
            'xinhuanet.com': '新华网',
        }
        for key, val in mapping.items():
            if key in domain:
                return val
        return domain.replace('www.', '').split('.')[0]
    return ''

def main():
    mmdd = (datetime.now() + timedelta(days=1)).strftime("%m%d")
    origin_path = os.path.join(ORIGIN_DIR, f"{mmdd}data.json")
    if not os.path.exists(origin_path):
        mmdd = datetime.now().strftime("%m%d")
        origin_path = os.path.join(ORIGIN_DIR, f"{mmdd}data.json")
    print(f"[{datetime.now().strftime('%H:%M:%S')}] 抓取社会热点")
    print(f"  origin: {origin_path}")

    today_str = datetime.now().strftime("%Y-%m-%d")

    # 抓取3页
    all_articles = []
    seen_titles = set()
    for p in range(1, 4):
        items = fetch_page(p)
        for item in items:
            t = item['title']
            if t not in seen_titles:
                seen_titles.add(t)
                # 按social_hotspots结构组装
                entry = {
                    "title": t,
                    "summary": "",
                    "focus_point": "",
                    "publish_time": today_str,
                    "thumb": "",
                    "source_url": item['url'],
                    "platform": get_domain(item['url']),
                    "heat_score":90,
                    "yesorno":""
                }
                all_articles.append(entry)

    print(f"\n  去重后共 {len(all_articles)} 条")

    # 读取origin_data
    mmd_data = []
    existing_titles = set()
    if os.path.exists(origin_path):
        with open(origin_path, 'r', encoding='utf-8') as f:
            mmd_data = json.load(f)
        # 收集已有social_hotspots的title
        for entry in mmd_data:
            if entry.get('name') == 'social_hotspots':
                for it in entry.get('list', []):
                    t = it.get('title', '')
                    if t:
                        existing_titles.add(t)

    # 插入到已有数据前边（去重）
    new_items = [a for a in all_articles if a['title'] not in existing_titles]
    print(f"  新增 {len(new_items)} 条，跳过 {len(all_articles)-len(new_items)} 条重复")

    if new_items:
        # prepend
        for entry in mmd_data:
            if entry.get('name') == 'social_hotspots':
                entry['list'] = new_items + entry.get('list', [])
                break
        else:
            mmd_data.append({"name": "social_hotspots", "opinion": "", "list": new_items})

    # 保存
    os.makedirs(ORIGIN_DIR, exist_ok=True)
    with open(origin_path, 'w', encoding='utf-8') as f:
        json.dump(mmd_data, f, ensure_ascii=False, indent=2)

    total = 0
    for entry in mmd_data:
        if entry.get('name') == 'social_hotspots':
            total = len(entry.get('list', []))
    print(f"\n✅ 已写入 {origin_path}（social_hotspots共{total}条）")

if __name__ == '__main__':
    main()
