#!/usr/bin/python3.12
"""从 chooseauto.com.cn 抓取今日车型新闻，追加到 origin_data/{MMDD}data.json 的 vehicle_hotspots"""
import urllib.request, re, json, os, sys, time
from datetime import datetime
from lxml import html as lxml_html


# 车型名关键词（有这些才算有具体车型）
_MODEL_KEYWORDS = [
    'GT', 'GTi', 'EV', 'SUV', 'MPV', 'PHEV', 'DM-i', 'L', 'Pro', 'Plus', 'MAX',
    '纯电', '插混', '混动', '增程', '电动',
    '款', '换代', '改款', '新款', '全新',
]
_MODEL_PATTERN = __import__('re').compile(r'[A-Z0-9]{2,}')


def _has_model(title):
    """检查标题是否有具体车型名"""
    # 包含数字+字母组合（如 Q7、X5、Z20、K50、i60、G700 等）
    if _MODEL_PATTERN.search(title):
        return True
    # 包含车型关键词
    for kw in _MODEL_KEYWORDS:
        if kw in title:
            return True
    # 包含常见车型品牌后缀
    for suffix in ['版', '型', '代', '系']:
        for num in ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9']:
            if num in title:
                return True
    return False

MMDD = datetime.now().strftime('%m%d')
ORIGIN_PATH = f'/data/news/json/origin_data/{MMDD}data.json'

URLS = [
    'https://www.chooseauto.com.cn/list/channel_1.shtml',
    'https://www.chooseauto.com.cn/list/channel_1_2.html',
]


def decode_resp(resp):
    raw = resp.read()
    ct = resp.headers.get('Content-Type', '')
    m = re.search(r'charset=([\w-]+)', ct, re.I)
    if m:
        try:
            return raw.decode(m.group(1).lower())
        except:
            pass
    m2 = re.search(b'charset=["\s]*([\\w-]+)', raw[:2000], re.I)
    enc = m2.group(1).decode().lower() if m2 else 'utf-8'
    try:
        return raw.decode(enc)
    except:
        return raw.decode('utf-8', errors='replace')


def fetch_list(url):
    try:
        req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        html = decode_resp(resp)
    except Exception as e:
        print(f'    W: {e}')
        return []

    items = []
    seen = set()
    for m in re.finditer(r'<a\s+href="(/news/\d+\.shtml)"[^>]*>([^<]+)</a>', html):
        href = 'https://www.chooseauto.com.cn' + m.group(1)
        title = m.group(2).strip()
        if title and len(title) > 4 and href not in seen:
            seen.add(href)
            items.append({'title': title, 'source_url': href})
    return items


def fetch_detail(item):
    try:
        req = urllib.request.Request(item['source_url'], headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        html = decode_resp(resp)
        tree = lxml_html.fromstring(html)

        thumb = ''
        for img in tree.xpath('//img[@src]'):
            src = img.get('src', '')
            if 'uploads' in src or 'chooseauto.com.cn' in src:
                if not src.startswith('http'):
                    src = 'https://www.chooseauto.com.cn' + src
                thumb = src
                break

        summary = ''
        meta = tree.xpath('//meta[@name="description"]/@content')
        if meta:
            summary = meta[0].strip()[:200]

        item['summary'] = summary
        item['thumb'] = thumb
    except:
        pass
    return item


def merge_to_origin(articles):
    if os.path.exists(ORIGIN_PATH):
        with open(ORIGIN_PATH, encoding='utf-8') as f:
            data = json.load(f)
    else:
        data = []

    date_str = datetime.now().strftime('%Y-%m-%d')
    for entry in data:
        if entry.get('name') in ('vehicle_hotspots', 'vehicle_hotposts'):
            existing = {e.get('title', '') for e in entry.get('list', [])}
            added = 0
            for a in articles:
                if a['title'] not in existing:
                    entry['list'].append({
                        'model': '', 'title': a['title'],
                        'summary': a.get('summary', ''), 'focus_point': '',
                        'publish_time': date_str, 'thumb': a.get('thumb', ''),
                        'source_url': a.get('source_url', ''),
                        'platform': '选车网',
                        'origin_url': '', 'yesorno': '',
                    })
                    existing.add(a['title'])
                    added += 1
            print(f'  新增 {added} 条（跳过 {len(articles)-added} 重复）')
            os.makedirs(os.path.dirname(ORIGIN_PATH), exist_ok=True)
            with open(ORIGIN_PATH, 'w', encoding='utf-8') as f:
                json.dump(data, f, ensure_ascii=False, indent=2)
            return

    data.append({
        'name': 'vehicle_hotspots', 'opinion': '',
        'list': [{
            'model': '', 'title': a['title'], 'summary': a.get('summary', ''),
            'focus_point': '', 'publish_time': date_str, 'thumb': a.get('thumb', ''),
            'source_url': a.get('source_url', ''), 'platform': '选车网',
            'origin_url': '', 'yesorno': '',
        } for a in articles]
    })
    with open(ORIGIN_PATH, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print(f'  新建 vehicle_hotspots, 新增 {len(articles)} 条')


def main():
    print(f'[{datetime.now().strftime("%H:%M:%S")}] 抓取 chooseauto...')
    all_items = []
    seen = set()

    for url in URLS:
        items = fetch_list(url)
        for it in items:
            if it['title'] not in seen:
                seen.add(it['title'])
                all_items.append(it)
        print(f'  {url.split("/")[-1]}: {len(items)} 条')
        time.sleep(0.3)

    print(f'\n  共 {len(all_items)} 条，抓取详情...')
    for i, it in enumerate(all_items):
        fetch_detail(it)
        print(f'    [{i+1}/{len(all_items)}] {it["title"][:30]}')
        time.sleep(0.3)

    all_items = [it for it in all_items if _has_model(it.get('title', ''))]
    if all_items:
        merge_to_origin(all_items)
    print(f'\n完成')


if __name__ == '__main__':
    main()
