#!/usr/bin/python3.12
"""从 itopmarketing.com 抓取文章，追加到 origin_data/{MMDD}data.json 的 brand_hotspots"""
import urllib.request, re, json, sys, os, time
from datetime import datetime, timedelta
from lxml import html as lxml_html

URL = 'https://itopmarketing.com/'
DAYS_BACK = 2


def fetch_articles():
    """获取首页文章列表 + 详情摘要"""
    req = urllib.request.Request(URL, headers={'User-Agent': 'Mozilla/5.0'})
    resp = urllib.request.urlopen(req, timeout=15)
    tree = lxml_html.fromstring(resp.read())

    articles = []
    seen_urls = set()
    cutoff_date = (datetime.now() - timedelta(days=DAYS_BACK)).strftime('%Y-%m-%d')

    for a in tree.xpath('//a[contains(@href, "/info")]'):
        href = a.get('href', '')
        if not href.startswith('http'):
            href = 'https://itopmarketing.com' + href
        title = a.text_content().strip()
        if not title or len(title) < 4 or href in seen_urls:
            continue
        seen_urls.add(href)

        parent = a
        img_src = ''
        for _ in range(5):
            imgs = parent.xpath('.//img')
            if imgs:
                img_src = imgs[0].get('src', '')
                break
            p = parent.getparent()
            if p is not None:
                parent = p
            else:
                break
        if img_src and not img_src.startswith('http'):
            img_src = 'https://itopmarketing.com' + img_src

        pub_date = ''
        date_m = re.search(r'/uploads/image/(\d{4})(\d{2})(\d{2})/', img_src)
        if date_m:
            pub_date = f'{date_m.group(1)}-{date_m.group(2)}-{date_m.group(3)}'

        title = title.replace('\u201c', '"').replace('\u201d', '"').replace('\u2014', '—')
        if '著）' in title or '著)' in title or pub_date < cutoff_date:
            continue

        articles.append({
            'title': title, 'summary': '', 'focus_point': '',
            'publish_time': pub_date, 'thumb': img_src,
            'source_url': href, 'platform': 'itopmarketing', 'brand': ''
        })

    # 抓取摘要
    for i, art in enumerate(articles):
        try:
            req = urllib.request.Request(art['source_url'], headers={'User-Agent': 'Mozilla/5.0'})
            resp = urllib.request.urlopen(req, timeout=15)
            page = lxml_html.fromstring(resp.read())

            zong = page.xpath('//*[contains(@class, "main_zong")]')
            if zong:
                for p in zong[0].xpath('.//p'):
                    text = p.text_content().strip()
                    if any(kw in text for kw in ['声明', '题图', '转载', '原创文章', '编辑', '作者']):
                        continue
                    if len(text) > 20:
                        art['summary'] = text[:200]
                        break
            print(f'  [{i + 1}/{len(articles)}] {art["title"][:30]}')
            time.sleep(0.3)
        except Exception as e:
            print(f'  [{i + 1}/{len(articles)}] ❌ {str(e)[:30]}')

    return articles


def merge_to_origin(new_articles):
    """追加到 origin_data/{MMDD}data.json 的 brand_hotspots"""
    mmdd = datetime.now().strftime('%m%d')
    origin_path = f'/data/news/json/origin_data/{mmdd}data.json'

    if os.path.exists(origin_path):
        with open(origin_path, 'r', encoding='utf-8') as f:
            data = json.load(f)
    else:
        print(f'  ⚠️ {origin_path} 不存在，新建')
        data = []

    # 找到 brand_hotspots
    found = False
    for entry in data:
        if entry.get('name') == 'brand_hotspots':
            existing_titles = {e.get('title', '') for e in entry.get('list', [])}
            added = 0
            for art in new_articles:
                if art['title'] not in existing_titles:
                    # 补充缺少的字段
                    item = {
                        'brand': art.get('brand', ''),
                        'title': art['title'],
                        'summary': art.get('summary', ''),
                        'focus_point': '',
                        'source_url': art.get('source_url', ''),
                        'publish_time': art.get('publish_time', ''),
                        'platform': 'itopmarketing',
                        'thumb': art.get('thumb', ''),
                        'origin_url': '',
                        'yesorno': '',
                    }
                    entry['list'].append(item)
                    existing_titles.add(art['title'])
                    added += 1
            print(f'\n  brand_hotspots 新增 {added} 条（跳过 {len(new_articles) - added} 条重复）')
            found = True
            break

    if not found:
        data.append({
            'name': 'brand_hotspots',
            'opinion': '',
            'list': [{
                'brand': a.get('brand', ''),
                'title': a['title'],
                'summary': a.get('summary', ''),
                'focus_point': '',
                'source_url': a.get('source_url', ''),
                'publish_time': a.get('publish_time', ''),
                'platform': 'itopmarketing',
                'thumb': a.get('thumb', ''),
                'origin_url': '',
                'yesorno': '',
            } for a in new_articles]
        })
        print(f'\n  新建 brand_hotspots，新增 {len(new_articles)} 条')

    with open(origin_path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

    print(f'  ✅ 已保存: {origin_path}')
    return origin_path


BRAND_LIST = ['亨氏','亚朵','苹果','百威','DIESEL','Tinder','Burberry','华莱士','NIKE','耐克',
               'adidas','阿迪达斯','可口可乐','百事','乐事','瑞幸','库迪','LV','Gucci','爱马仕',
               '卡地亚','三星','美的','伊利','蒙牛','特仑苏','王老吉','快手','公牛','长安马自达',
               '库里','LOEWE','老乡鸡','MARSHALL','ubras','黄仁勋','康师傅','美团','小红书',
               '快手','番茄小说','Heaven&Hell','Plaud','微信','番茄',
               '宝马','奔驰','奥迪','大众','丰田','本田','日产','比亚迪','小米','蔚来','理想',
               '小鹏','特斯拉','吉利','奇瑞','长城','长安','红旗','五菱','现代','起亚','捷尼赛思']


def has_brand(title):
    """检查标题是否包含已知品牌名"""
    for b in BRAND_LIST:
        if b in title:
            return True
    return False


def main():
    mmdd = datetime.now().strftime('%m%d')
    print(f'[{datetime.now().strftime("%H:%M:%S")}] 获取 itopmarketing...')

    articles = fetch_articles()
    
    # 过滤：只保留包含品牌名的内容
    before = len(articles)
    articles = [a for a in articles if has_brand(a['title'])]
    after = len(articles)
    if before - after > 0:
        print(f'  过滤 {before - after} 条无品牌内容（保留 {after} 条）')
    
    print(f'  共 {len(articles)} 篇')

    # 日期过滤：只保留今天和昨天
    today = datetime.now().strftime('%Y-%m-%d')
    yesterday = (datetime.now() - timedelta(days=1)).strftime('%Y-%m-%d')
    before = len(articles)
    articles = [a for a in articles if a.get('publish_time', '') in (today, yesterday) or not a.get('publish_time', '')]
    after = len(articles)
    if before - after > 0:
        print(f'  日期过滤: 移除 {before - after} 条非24h数据')

    if articles:
        merge_to_origin(articles)

    print(f'\n[{datetime.now().strftime("%H:%M:%S")}] 完成')


if __name__ == '__main__':
    main()
