#!/usr/bin/python3.12
"""从汽车之家(autohome)和易车(yiche/bitauto)抓取今日车型新闻，追加到 origin_data/{MMDD}data.json 的 vehicle_hotspots"""

import json, os, sys, re, signal
from datetime import datetime

# ── 车型名关键词（复用现有脚本规则） ──────────────────────────────────
_MODEL_KEYWORDS = [
    'GT', 'GTi', 'EV', 'SUV', 'MPV', 'PHEV', 'DM-i', 'L', 'Pro', 'Plus', 'MAX',
    '纯电', '插混', '混动', '增程', '电动',
    '款', '换代', '改款', '新款', '全新',
]
_MODEL_PATTERN = re.compile(r'[A-Z0-9]{2,}')

def _has_model(title):
    """检查标题是否有具体车型名"""
    if _MODEL_PATTERN.search(title):
        return True
    for kw in _MODEL_KEYWORDS:
        if kw in title:
            return True
    for suffix in ['版', '型', '代', '系']:
        for num in '0123456789':
            if num in title:
                return True
    return False

# ── 路径常量 ─────────────────────────────────────────────────────
TODAY = datetime.now()
MMDD = TODAY.strftime('%m%d')
ORIGIN_PATH = f'/data/news/json/origin_data/{MMDD}data.json'

# ── 工具：读取/初始化 origin_data ────────────────────────────────
def load_origin():
    if os.path.exists(ORIGIN_PATH):
        with open(ORIGIN_PATH, 'r', encoding='utf-8') as f:
            return json.load(f)
    return []

def save_origin(data):
    os.makedirs(os.path.dirname(ORIGIN_PATH), exist_ok=True)
    with open(ORIGIN_PATH, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

def merge_vehicle_list(origin_data, new_items, seen_titles):
    """合并新条目到vehicle_hotspots，去重"""
    for section in origin_data:
        if section.get('name') == 'vehicle_hotspots':
            existing = section['list']
            for item in new_items:
                t = item.get('title', '')
                if t and t not in seen_titles:
                    seen_titles.add(t)
                    existing.append(item)
            return origin_data
    # 没有vehicle_hotspots段，新建
    origin_data.append({
        "name": "vehicle_hotspots",
        "opinion": "",
        "list": new_items
    })
    return origin_data

def ensure_fields(item, platform):
    """补充缺失字段"""
    item.setdefault('focus_point', '')
    item.setdefault('yesorno', '')
    item.setdefault('origin_url', item.get('source_url', ''))
    item['platform'] = platform
    return item

# ═══════════════════════════════════════════════════════════════
#  汽车之家 (requests + 正则，SSR HTML)
# ═══════════════════════════════════════════════════════════════
def scrape_autohome():
    """从 autohome.com.cn/news/ 抓取（无需浏览器渲染）"""
    import requests as _req
    url = 'https://www.autohome.com.cn/news/'
    headers = {
        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
        'Referer': 'https://www.autohome.com.cn/',
    }
    try:
        resp = _req.get(url, headers=headers, timeout=20)
        resp.raise_for_status()
    except Exception as e:
        print(f'  ⚠️ autohome 请求失败: {e}', file=sys.stderr)
        return []

    text = resp.content.decode('gbk', errors='replace')

    # 提取所有 article 条目
    entries = re.findall(
        r'<li[^>]*data-artidanchor="(\d+)"[^>]*>.*?'
        r'<a[^>]*href="(//www\.autohome\.com\.cn/news/\d{6}/\d+[^"]*)"[^>]*>.*?'
        r'<img[^>]*src="([^"]*)"[^>]*>.*?'
        r'<h[23]>(.*?)</h[23]>.*?'
        r'fn-left[^>]*>(.*?)<',
        text, re.DOTALL
    )
    print(f'  autohome raw entries: {len(entries)}')

    # 提取摘要
    summaries = re.findall(
        r'data-artidanchor="(\d+)"[^>]*>.*?<p>(.*?)</p>',
        text, re.DOTALL
    )
    summary_map = {aid: re.sub(r'<[^>]+>', '', s).strip() for aid, s in summaries}

    items = []
    for aid, url, img, title_html, time_text in entries:
        title = re.sub(r'<[^>]+>', '', title_html).strip()
        if not title or len(title) < 5:
            continue
        summary = summary_map.get(aid, '')
        thumb = ('https:' + img) if not img.startswith('http') else img
        link = 'https:' + url.split('#')[0]

        # 发布时间：相对时间转日期
        pub = time_text.strip()
        now = datetime.now()
        if '分钟' in pub:
            mins = int(re.search(r'(\d+)', pub).group(1))
            pub_date = now.strftime('%Y-%m-%d')
        elif '小时' in pub:
            hrs = int(re.search(r'(\d+)', pub).group(1))
            from datetime import timedelta
            pub_date = (now - timedelta(hours=hrs)).strftime('%Y-%m-%d')
        elif '天前' in pub:
            days = int(re.search(r'(\d+)', pub).group(1))
            from datetime import timedelta
            pub_date = (now - timedelta(days=days)).strftime('%Y-%m-%d')
        else:
            pub_date = now.strftime('%Y-%m-%d')

        items.append(ensure_fields({
            'model': extract_model_name(title),
            'title': title,
            'summary': summary,
            'publish_time': pub_date,
            'thumb': thumb,
            'source_url': link,
        }, '汽车之家'))

    # 过滤不含车型的条目
    filtered = [i for i in items if _has_model(i['title'])]
    # 只保留今天的数据（昨天的数据由评分阶段的24h窗口处理）
    today_str = TODAY.strftime('%Y-%m-%d')
    filtered = [i for i in filtered if i.get('publish_time','')[:10] == today_str]
    print(f'  autohome -> {len(filtered)}条（含车型，共{len(items)}条原始）')
    return filtered

def extract_model_name(title):
    """从标题中提取车型名（品牌+型号），如'理想L6'、'小鹏MONA L03'"""
    # 品牌列表（按长度降序）
    brands = sorted([
        '凯迪拉克','劳斯莱斯','兰博基尼','玛莎拉蒂','保时捷','帕加尼','Hennessey',
        '捷尼赛思','Polestar','阿尔法·罗密欧','阿尔法罗密欧','北京越野','北京汽车',
        '沃尔沃','特斯拉','路特斯','雪铁龙','斯巴鲁','三菱','捷豹','路虎',
        '丰田','本田','日产','现代','起亚','福特','马自达','宝马','奔驰','奥迪',
        '比亚迪','吉利','长城','奇瑞','长安','五菱','别克','大众','标致',
        '理想','蔚来','小鹏','小米','极氪','问界','智界','享界','阿维塔',
        '零跑','哪吒','岚图','腾势','极狐','智己','飞凡','埃安','昊铂',
        '方程豹','仰望','坦克','魏牌','捷途','星途','林肯','smart','MINI',
        'Jeep','奔驰AMG','AMG','迈巴赫','杜卡迪','Zenvo','丰田GR','GR',
        '大众ID','ID',
    ], key=len, reverse=True)

    # 先尝试找 "品牌+空格?+型号" 的模式：型号至少包含一个数字或大写字母簇
    for b in brands:
        if b not in title:
            continue
        rest = title.split(b, 1)[1].strip()
        # 从 rest 头部提取型号字符（字母、数字、空格、点、斜杠、横线），遇到动词/分隔符/中文停止
        model_chars = []
        for ch in rest:
            if ch in '将于发布亮相曝光官图上市预售开启首发亮相售月日':
                break
            if ch in '（）()\r\n《》':
                break
            if ch in '，。、？！：；':
                break
            if '\u4e00' <= ch <= '\u9fff' and ch not in 'X':
                # 中文非数字类停止，但允许中文"中国红"之类的附加描述
                if model_chars and any('\u4e00' <= c <= '\u9fff' for c in model_chars):
                    # 已经有中文了，继续收集中文
                    model_chars.append(ch)
                elif not model_chars:
                    # 开头就是中文，不收（品牌名后直接跟中文描述）
                    break
                else:
                    # 英文数字后面遇到中文，停止
                    break
            else:
                model_chars.append(ch)
        model_str = ''.join(model_chars).strip(' \t\r\n.-')

        # 检查 model_str 是否包含有效的型号标识（数字或至少3个大写字母）
        if re.search(r'\d', model_str):
            return (b + ' ' + model_str)[:40]
        if re.search(r'[A-Z]{3,}', model_str):
            return (b + ' ' + model_str)[:40]

        # 否则尝试从整段 rest 中找 数字+字母 组合
        m = re.search(r'([A-Z0-9][A-Z0-9\s./-]{0,12}\d)', rest)
        if m:
            return (b + ' ' + m.group(1).strip())[:40]
        # 最后 fallback：品牌+后取到第一个中文前
        short = ''
        for ch in model_str:
            if '\u4e00' <= ch <= '\u9fff':
                break
            short += ch
        short = short.strip()
        if short:
            return (b + ' ' + short)[:40]
        return b[:40]

    # 无品牌匹配，找字母数字组合（纯英文型号）
    m = re.search(r'([A-Z][A-Za-z0-9]{1,15}(?:\s[A-Z0-9]{1,5})?)', title)
    if m:
        return m.group(1)[:30]

    # 最后 fallback：常见中文车型名（无品牌前缀）
    model_names = ['飞度','思域','雅阁','凯美瑞','卡罗拉','汉兰达','霸道','陆巡',
                   '大汉','澎程','坦克300','坦克500','豹5','豹8','宋','秦','汉','唐',
                   '元','海豹','海豚','海狮','海鸥','驱逐舰','护卫舰','银河TT',
                   '星舰7','星耀7','L6','L7','L8','L9','M7','M9','MEGA','SU7',
                   'YU7','X5','X7','Q5','Q7','GLC','GLE','GLS','EQC','EQS','iX3',
                   'i4','i5','i7','iX','3系','5系','7系','X1','X3','X5']
    for n in model_names:
        if n in title:
            return n[:30]

    return ''

# ═══════════════════════════════════════════════════════════════
#  易车 / Bitauto (Playwright 渲染，JS页面)
# ═══════════════════════════════════════════════════════════════
def scrape_yiche():
    """从 www.yiche.com/news/ 跳转到 bitauto.com 抓取"""
    from playwright.sync_api import sync_playwright
    items = []
    try:
        with sync_playwright() as pw:
            browser = pw.chromium.launch(
                headless=True,
                args=['--no-sandbox', '--disable-gpu', '--disable-dev-shm-usage']
            )
            page = browser.new_page(viewport={'width': 1440, 'height': 900})
            page.goto('https://www.yiche.com/news/', timeout=60000, wait_until='networkidle')
            page.wait_for_timeout(3000)

            # 提取所有文章
            articles = page.evaluate('''() => {
                const items = [];
                document.querySelectorAll('li').forEach(li => {
                    const link = li.querySelector('a[href*="/news/"]');
                    if (!link) return;
                    const img = link.querySelector('img');
                    const titleDiv = li.querySelector('.item-title');
                    const summaryDiv = li.querySelector('.item-summary');
                    if (titleDiv) {
                        items.push({
                            title: titleDiv.innerText.trim(),
                            summary: summaryDiv ? summaryDiv.innerText.trim() : '',
                            href: 'https://www.bitauto.com' + link.getAttribute('href'),
                            img: img ? img.src : ''
                        });
                    }
                });
                return items;
            }''')

            for a in articles:
                title = a['title']
                if not title or len(title) < 5:
                    continue
                # 从图片URL提取日期
                img_url = a['img']
                m = re.search(r'/2026(\d{4})/', img_url)
                pub_date = f"2026-{m.group(1)[:2]}-{m.group(1)[2:]}" if m else TODAY.strftime('%Y-%m-%d')

                items.append(ensure_fields({
                    'model': extract_model_name(title),
                    'title': title,
                    'summary': a['summary'],
                    'publish_time': pub_date,
                    'thumb': img_url,
                    'source_url': a['href'],
                }, '易车'))

            browser.close()
    except Exception as e:
        print(f'  ⚠️ yiche 抓取失败: {e}', file=sys.stderr)
        return []

    filtered = [i for i in items if _has_model(i['title'])]
    # 只保留今天的数据
    today_str = TODAY.strftime('%Y-%m-%d')
    filtered = [i for i in filtered if i.get('publish_time','')[:10] == today_str]
    print(f'  yiche -> {len(filtered)}条（含车型，共{len(items)}条原始）')
    return filtered

# ═══════════════════════════════════════════════════════════════
#  主入口
# ═══════════════════════════════════════════════════════════════
def main():
    print(f'📅 日期: {TODAY.strftime("%Y-%m-%d")}, MMDD={MMDD}')
    print(f'📄 目标: {ORIGIN_PATH}')

    origin_data = load_origin()
    seen_titles = set()

    # 收集已有标题用于去重
    for section in origin_data:
        if section.get('name') == 'vehicle_hotspots':
            for item in section.get('list', []):
                t = item.get('title', '')
                if t:
                    seen_titles.add(t)

    total_new = 0

    # ── 1. 汽车之家 ──
    print('\n🔍 [1/2] 汽车之家 autohome.com.cn/news/')
    autohome_items = scrape_autohome()
    if autohome_items:
        origin_data = merge_vehicle_list(origin_data, autohome_items, seen_titles)
        total_new += len(autohome_items)
        save_origin(origin_data)
        print(f'  ✅ 已写入 {len(autohome_items)} 条')

    # ── 2. 易车/bitauto ──
    print('\n🔍 [2/2] 易车 www.yiche.com/news/ (Playwright)')
    yiche_items = scrape_yiche()
    if yiche_items:
        origin_data = merge_vehicle_list(origin_data, yiche_items, seen_titles)
        total_new += len(yiche_items)
        save_origin(origin_data)
        print(f'  ✅ 已写入 {len(yiche_items)} 条')

    # 最终统计
    for section in origin_data:
        if section.get('name') == 'vehicle_hotspots':
            total = len(section['list'])
            print(f'\n📊 汇总: vehicle_hotspots 共 {total} 条，本次新增 {total_new} 条')
            break
    else:
        print('\n❌ vehicle_hotspots 未创建')

if __name__ == '__main__':
    main()
