#!/usr/bin/env python3
"""通过source_url域名匹配媒体名称，更新platform字段"""
import json, os, sys
from urllib.parse import urlparse

# 媒体域名映射表（可维护）
MEDIA_LIST = [
    { "name": "36氪", "host": "36kr.com" },
  { "name": "BBX", "host": "bbx.cash" },
  { "name": "百度", "host": "baidu.com" },
  { "name": "巴比特", "host": "btcfans.com" },
  { "name": "币看", "host": "bitkan.com" },
  { "name": "哔哩哔哩", "host": "bilibili.com" },
  { "name": "链捕手", "host": "chaincatcher.com" },
  { "name": "新浪财经", "host": "cj.sina.cn" },
  { "name": "新浪财经", "host": "cj.sina.com.cn" },
  { "name": "Followin", "host": "followin.io" },
  { "name": "格隆汇", "host": "gelonghui.com" },
  { "name": "观点网", "host": "guandian.cn" },
  { "name": "虎嗅", "host": "huxiu.com" },
  { "name": "韩联社", "host": "yna.co.kr" },
  { "name": "ZAKER", "host": "myzaker.com" },
  { "name": "ODaily", "host": "odaily.news" },
  { "name": "PANews", "host": "panewslab.com" },
  { "name": "美通社", "host": "prnasia.com" },
  { "name": "新浪财经", "host": "finance.sina.com.cn" },
  { "name": "深潮 TechFlow", "host": "techflowpost.com" },
  { "name": "钛媒体", "host": "tmtpost.com" },
  { "name": "今日头条", "host": "toutiao.com" },
  { "name": "IT之家", "host": "ithome.com" },
  { "name": "澎湃新闻", "host": "thepaper.cn" },
  { "name": "观察者网", "host": "guancha.cn" },
  { "name": "虎嗅", "host": "huxiucdn.com" },
  { "name": "界面新闻", "host": "jiemian.com" },
  { "name": "华尔街见闻", "host": "wallstreetcn.com" },
  { "name": "新浪财经", "host": "sina.com.cn" },
  { "name": "新浪", "host": "sina.cn" },
  { "name": "网易", "host": "163.com" },
  { "name": "搜狐", "host": "sohu.com" },
  { "name": "腾讯新闻", "host": "news.qq.com" },
  { "name": "腾讯新闻", "host": "view.inews.qq.com" },
  { "name": "新华网", "host": "news.cn" },
  { "name": "新华网", "host": "xinhuanet.com" },
  { "name": "人民网", "host": "people.com.cn" },
  { "name": "央视新闻", "host": "cctv.com" },
  { "name": "央视网", "host": "news.cctv.com" },
  { "name": "第一财经", "host": "yicai.com" },
  { "name": "第一财经", "host": "m.yicai.com" },
  { "name": "每日经济新闻", "host": "nbd.com.cn" },
  { "name": "经济观察报", "host": "eeo.com.cn" },
  { "name": "21世纪经济报道", "host": "21jingji.com" },
  { "name": "中国证券报", "host": "cs.com.cn" },
  { "name": "证券时报", "host": "stcn.com" },
  { "name": "上海证券报", "host": "cnstock.com" },
  { "name": "中国基金报", "host": "chinafundnews.com" },
  { "name": "财联社", "host": "cls.cn" },
  { "name": "北京日报", "host": "bjd.com.cn" },
  { "name": "北京日报", "host": "bjnews.com.cn" },
  { "name": "上观新闻", "host": "shobserver.com" },
  { "name": "中国新闻网", "host": "chinanews.com" },
  { "name": "光明网", "host": "gmw.cn" },
  { "name": "中国经营报", "host": "cb.com.cn" },
  { "name": "国际金融报", "host": "gfic.cn" },
  { "name": "环球网", "host": "huanqiu.com" },
  { "name": "参考消息", "host": "cankaoxiaoxi.com" },
  { "name": "中国青年报", "host": "cyol.net" },
  { "name": "今日头条", "host": "toutiaoimg.cn" },
  { "name": "微信公众号", "host": "mp.weixin.qq.com" },
  { "name": "知乎", "host": "zhihu.com" },
  { "name": "微博", "host": "weibo.com" },
  { "name": "微博", "host": "weibo.cn" },
  { "name": "百家号", "host": "baijiahao.baidu.com" },
  { "name": "汽车之家", "host": "autohome.com.cn" },
  { "name": "易车", "host": "yiche.com" },
  { "name": "盖世汽车", "host": "gasgoo.com" },
  { "name": "选车网", "host": "chooseauto.com.cn" },
  { "name": "第一电动", "host": "d1ev.com" },
  { "name": "懂车帝", "host": "dongchedi.com" },
  { "name": "SocialBeta", "host": "socialbeta.com" },
  { "name": "数英", "host": "digitaling.com" },
  { "name": "梅花网", "host": "meihua.info" },
  { "name": "品牌星球", "host": "brandstar.com.cn" },
  { "name": "广告门", "host": "adquan.com" },
  { "name": "iTop", "host": "itopmarketing.com" },
  { "name": "雷锋网", "host": "leiphone.com" },
  { "name": "CNMO", "host": "smartcar.cnmo.com" },
  { "name": "虎扑", "host": "bbs.hupu.com" }
]

def find_media(url):
    """从URL中匹配媒体名称"""
    if not url:
        return None
    try:
        domain = urlparse(url).netloc.lower()
        # 去掉www.前缀
        if domain.startswith('www.'):
            domain = domain[4:]
        for media in MEDIA_LIST:
            host = media['host'].lower()
            if host == domain or domain.endswith('.' + host):
                return media['name']
            # 也匹配子域名包含的情况
            if host in domain:
                return media['name']
    except:
        pass
    return None

def process_items(items, path_desc):
    """递归处理list中的条目"""
    count = 0; fixed = 0
    for item in items:
        if isinstance(item, dict):
            url = item.get('source_url', '') or item.get('url', '')
            if url:
                name = find_media(url)
                if name:
                    old = item.get('platform', '')
                    if old != name:
                        item['platform'] = name
                        count += 1
                t = item.get('title', '') or ''
                if t and name:
                    import re as _re
                    _new_t = _re.sub(r'[-_—][\s]*' + _re.escape(name) + r'$', '', t).strip()
                    if _new_t != t:
                        item['title'] = _new_t
                        fixed += 1
        elif isinstance(item, list):
            sub_cnt, sub_fix = process_items(item, path_desc)
            count += sub_cnt; fixed += sub_fix
    return count, fixed

def main():
    if len(sys.argv) < 2:
        print("用法: python3 get_media_by_sourceurl.py <json文件路径>")
        sys.exit(1)

    path = sys.argv[1]
    if not os.path.exists(path):
        print(f"文件不存在: {path}")
        sys.exit(1)

    with open(path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    total = 0
    title_fixed = 0

    if isinstance(data, list):
        for item in data:
            if isinstance(item, dict) and 'list' in item:
                cnt, fix = process_items(item["list"], item.get("name", ""))
                total += cnt; title_fixed += fix
            elif isinstance(item, dict):
                # 可能是直接的对象
                for k, v in item.items():
                    if isinstance(v, list):
                        cnt, fix = process_items(v, k)
                total += cnt; title_fixed += fix
    elif isinstance(data, dict):
        for k, v in data.items():
            if isinstance(v, list):
                cnt, fix = process_items(v, k)
                total += cnt; title_fixed += fix

    with open(path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

    print(f"✅ 更新了 {total} 条 platform 字段")
    if title_fixed:
        print(f"   ✅ 清理了 {title_fixed} 条标题后缀")
    if title_fixed:
        print(f"   ✅ 清理了 {title_fixed} 条标题后缀")
    print(f"   已保存到 {path}")

if __name__ == '__main__':
    main()
