#!/usr/bin/python3.12
"""从 12365auto.com 抓取今日车型新闻，追加到 origin_data/{MMDD}data.json 的 vehicle_hotspots"""
import urllib.request, re, json, os, sys, time
from datetime import datetime
from lxml import html as lxml_html


# 车型名关键词（有这些才算有具体车型）
_MODEL_KEYWORDS = [
    'GT', 'GTi', 'EV', 'SUV', 'MPV', 'PHEV', 'DM-i', 'L', 'Pro', 'Plus', 'MAX',
    '纯电', '插混', '混动', '增程', '电动',
    '款', '换代', '改款', '新款', '全新',
]
_MODEL_PATTERN = __import__('re').compile(r'[A-Z0-9]{2,}')


def _has_model(title):
    """检查标题是否有具体车型名"""
    # 包含数字+字母组合（如 Q7、X5、Z20、K50、i60、G700 等）
    if _MODEL_PATTERN.search(title):
        return True
    # 包含车型关键词
    for kw in _MODEL_KEYWORDS:
        if kw in title:
            return True
    # 包含常见车型品牌后缀
    for suffix in ['版', '型', '代', '系']:
        for num in ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9']:
            if num in title:
                return True
    return False

TODAY = datetime.now().strftime('%Y%m%d')
MMDD = datetime.now().strftime('%m%d')
ORIGIN_PATH = f'/data/news/json/origin_data/{MMDD}data.json'

URLS = [
    ('https://www.12365auto.com/xcdg/index.shtml', '车质网'),
    ('https://www.12365auto.com/xcdg/index_2.shtml', '车质网'),
    ('https://www.12365auto.com/news/index.shtml', '车质网'),
    ('https://www.12365auto.com/news/index_2.shtml', '车质网'),
]


def decode_resp(resp):
    """自动检测编码"""
    raw = resp.read()
    ct = resp.headers.get('Content-Type', '')
    m = re.search(r'charset=([\w-]+)', ct, re.I)
    if m:
        enc = m.group(1).lower()
        if enc == 'utf-8':
            return raw.decode('utf-8', errors='replace')
    # HTTP头没给编码时，从meta检测
    m2 = re.search(b'charset=[\\s"]*([\\w-]+)', raw[:2000], re.I)
    if m2:
        meta_enc = m2.group(1).decode().lower()
    else:
        meta_enc = 'utf-8'
    for enc in [meta_enc, 'gbk', 'utf-8']:
        try:
            return raw.decode(enc)
        except:
            continue
    return raw.decode('utf-8', errors='replace')


def fetch_list(url, source_name):
    """正则提取列表页的文章标题和链接"""
    try:
        req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        html = decode_resp(resp)
    except Exception as e:
        print(f'    W: 请求失败 {e}')
        return []

    items = []
    seen = set()
    pattern = rf'<a\s+href="([^"]*{TODAY}[^"]*)"[^>]*>([^<]+)</a>'
    for m in re.finditer(pattern, html):
        href = m.group(1)
        title = m.group(2).strip()
        if not title or len(title) < 4 or href in seen:
            continue
        seen.add(href)
        if not href.startswith('http'):
            href = 'https://www.12365auto.com' + href
        items.append({'title': title, 'source_url': href, 'platform': source_name})
    return items


def fetch_detail(item):
    """详情页获取摘要和封面"""
    try:
        req = urllib.request.Request(item['source_url'], headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        html = decode_resp(resp)
        tree = lxml_html.fromstring(html)

        thumb = ''
        for img in tree.xpath('//img[@src]'):
            src = img.get('src', '')
            if any(kw in src for kw in ['uploads', '2026', 'topicNewsImage']):
                if not src.startswith('http'):
                    src = 'https://www.12365auto.com' + src
                thumb = src
                break

        summary = ''
        meta = tree.xpath('//meta[@name="description"]/@content')
        if meta:
            summary = meta[0].strip()[:200]
        if not summary:
            for div in tree.xpath('//*[contains(@class, "detail") or contains(@class, "content")]'):
                for p in div.xpath('.//p'):
                    t = p.text_content().strip()
                    if len(t) > 20:
                        summary = t[:200]
                        break
                if summary:
                    break

        item['summary'] = summary
        item['thumb'] = thumb
    except Exception as e:
        pass
    return item


def merge_to_origin(articles):
    if os.path.exists(ORIGIN_PATH):
        with open(ORIGIN_PATH, encoding='utf-8') as f:
            data = json.load(f)
    else:
        data = []

    date_str = f'{TODAY[:4]}-{TODAY[4:6]}-{TODAY[6:]}'
    for entry in data:
        if entry.get('name') in ('vehicle_hotspots', 'vehicle_hotposts'):
            existing = {e.get('title', '') for e in entry.get('list', [])}
            added = 0
            for a in articles:
                if a['title'] not in existing:
                    entry['list'].append({
                        'model': '', 'title': a['title'],
                        'summary': a.get('summary', ''), 'focus_point': '',
                        'publish_time': date_str, 'thumb': a.get('thumb', ''),
                        'source_url': a.get('source_url', ''),
                        'platform': a.get('platform', ''),
                        'origin_url': '', 'yesorno': '',
                    })
                    existing.add(a['title'])
                    added += 1
            print(f'  新增 {added} 条（跳过 {len(articles)-added} 重复）')
            os.makedirs(os.path.dirname(ORIGIN_PATH), exist_ok=True)
            with open(ORIGIN_PATH, 'w', encoding='utf-8') as f:
                json.dump(data, f, ensure_ascii=False, indent=2)
            print(f'  OK: {ORIGIN_PATH}')
            return

    # 没有vehicle_hotspots则新建
    data.append({
        'name': 'vehicle_hotspots', 'opinion': '',
        'list': [{'model': '', 'title': a['title'], 'summary': a.get('summary', ''),
                  'focus_point': '', 'publish_time': date_str, 'thumb': a.get('thumb', ''),
                  'source_url': a.get('source_url', ''), 'platform': a.get('platform', ''),
                  'origin_url': '', 'yesorno': ''} for a in articles]
    })
    os.makedirs(os.path.dirname(ORIGIN_PATH), exist_ok=True)
    with open(ORIGIN_PATH, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print(f'  新建 vehicle_hotspots, 新增 {len(articles)} 条')
    print(f'  OK: {ORIGIN_PATH}')


def main():
    print(f'[{datetime.now().strftime("%H:%M:%S")}] 抓取 12365auto...')
    all_items = []
    seen = set()

    for url, src in URLS:
        items = fetch_list(url, src)
        for it in items:
            if it['title'] not in seen:
                seen.add(it['title'])
                all_items.append(it)
        print(f'  {src}/{url.split("/")[-1]}: {len(items)} 条')
        time.sleep(0.3)

    print(f'\n  共 {len(all_items)} 条，抓取详情摘要...')
    for i, it in enumerate(all_items):
        fetch_detail(it)
        print(f'    [{i+1}/{len(all_items)}] {it["title"][:30]}')
        time.sleep(0.3)

    all_items = [it for it in all_items if _has_model(it.get('title', ''))]
    if all_items:
        merge_to_origin(all_items)
    print(f'\n完成')


if __name__ == '__main__':
    main()
