#!/usr/bin/env python3
"""
Icon 下载脚本
- JSON: 读取 icon 字段，下载外部链接到本地
- Markdown: 解析表格中"图标"列的链接，下载到本地
三级容错: 直接下载 -> 页面提取真实favicon -> 占位图

用法: python3 download_icons.py <file> [file2 ...]
"""

import json, os, ssl, urllib.request, urllib.parse, re, sys

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

ICONDIR = '/data/news/images/icon'
os.makedirs(ICONDIR, exist_ok=True)

PLACEHOLDER = bytes([0x89,0x50,0x4E,0x47,0x0D,0x0A,0x1A,0x0A,0x00,0x00,0x00,0x0D,0x49,0x48,0x44,0x52,0x00,0x00,0x00,0x01,0x00,0x00,0x00,0x01,0x08,0x06,0x00,0x00,0x00,0x1F,0x15,0xC4,0x89,0x00,0x00,0x00,0x0A,0x49,0x44,0x41,0x54,0x08,0xD7,0x63,0x68,0x60,0x00,0x00,0x00,0x02,0x00,0x01,0xE5,0x27,0xDE,0xFC,0x00,0x00,0x00,0x00,0x49,0x45,0x4E,0x44,0xAE,0x42,0x60,0x82])


def urlretrieve(url, timeout=10):
    try:
        req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0', 'Accept': 'image/*'})
        data = urllib.request.urlopen(req, timeout=timeout, context=ssl_ctx).read()
        return data if len(data) > 50 else None
    except:
        return None


def find_favicon_from_page(domain):
    for scheme in ['https', 'http']:
        try:
            req = urllib.request.Request(f'{scheme}://{domain}', headers={'User-Agent': 'Mozilla/5.0'})
            html = urllib.request.urlopen(req, timeout=8, context=ssl_ctx).read().decode('utf-8', 'ignore')
            patterns = [
                r'<link\s+[^>]*rel=["\'](?:shortcut\s+)?icon["\'][^>]*href=["\']([^"\']+)["\']',
                r'<link\s+[^>]*href=["\']([^"\']+)["\'][^>]*rel=["\'](?:shortcut\s+)?icon["\']'
            ]
            for pat in patterns:
                m = re.search(pat, html, re.IGNORECASE)
                if m:
                    fp = m.group(1)
                    if fp.startswith('//'):
                        return f'{scheme}:{fp}'
                    elif fp.startswith('/'):
                        return f'{scheme}://{domain}{fp}'
                    elif fp.startswith('http'):
                        return fp
                    else:
                        return f'{scheme}://{domain}/{fp}'
        except:
            continue
    return None


def safe_filename(url):
    parsed = urllib.parse.urlparse(url)
    domain = parsed.netloc.replace('www.', '').replace('.', '-')
    ext = os.path.splitext(parsed.path)[1]
    if ext not in ['.ico', '.png', '.jpg', '.jpeg', '.gif', '.svg']:
        ext = '.ico'
    return f'{domain}{ext}'


def download_icon(icon_url):
    fname = safe_filename(icon_url)
    fpath = os.path.join(ICONDIR, fname)

    data = urlretrieve(icon_url)
    if data:
        with open(fpath, 'wb') as f:
            f.write(data)
        return f'/images/icon/{fname}', f'OK: {fname} ({len(data)}B) direct'

    parsed = urllib.parse.urlparse(icon_url)
    domain = parsed.netloc or parsed.hostname
    if domain:
        real_url = find_favicon_from_page(domain)
        if real_url:
            data = urlretrieve(real_url)
            if data:
                with open(fpath, 'wb') as f:
                    f.write(data)
                return f'/images/icon/{fname}', f'OK: {fname} ({len(data)}B) extracted [{real_url}]'

    with open(fpath, 'wb') as f:
        f.write(PLACEHOLDER)
    return f'/images/icon/{fname}', f'FALLBACK: {fname} placeholder'


def extract_icons_from_markdown(text):
    """Parse markdown table, extract URLs from icon column."""
    lines = text.split('\n')
    icon_col = None
    for line in lines:
        cells = [c.strip() for c in line.split('|') if c.strip()]
        for i, cell in enumerate(cells):
            if '\u56fe\u6807' in cell:  # '图标'
                icon_col = i
                break
        if icon_col is not None:
            break

    if icon_col is None:
        print('  WARN: no icon column found')
        return []

    urls = set()
    for line in lines:
        cells = [c.strip() for c in line.split('|') if c.strip()]
        if len(cells) > icon_col:
            found = re.findall(r'https?://[^\s|]+', cells[icon_col])
            for url in found:
                if '.ico' in url.lower() or 'favicon' in url.lower():
                    urls.add(url)
    return list(urls)


def process_file(filepath):
    with open(filepath, 'r', encoding='utf-8') as f:
        content = f.read().strip()

    if content.startswith('{') or content.startswith('['):
        # JSON
        data = json.loads(content)
        updates = 0
        for sec in ['hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international', 'hyundai_buzz_topics']:
            for item in data.get(sec, []):
                ico = item.get('icon', '')
                if ico and (ico.startswith('http') or ico.startswith('//')):
                    local_path, msg = download_icon(ico)
                    item['icon'] = local_path
                    updates += 1
                    print(f'  {msg}')
        with open(filepath, 'w', encoding='utf-8') as f:
            json.dump(data, f, ensure_ascii=False, indent=2)
        print(f'\nDONE: {os.path.basename(filepath)}: {updates} icons processed')
    else:
        # Markdown
        urls = extract_icons_from_markdown(content)
        if not urls:
            print('  WARN: no icon URLs found in markdown')
            return
        print(f'  Found {len(urls)} icon URLs in markdown table')
        for url in urls:
            local_path, msg = download_icon(url)
            print(f'  {msg}')
        print(f'\nDONE: {os.path.basename(filepath)}: {len(urls)} icons downloaded')


if __name__ == '__main__':
    for path in sys.argv[1:]:
        if os.path.exists(path):
            print(f'\nFile: {path}')
            process_file(path)
        else:
            print(f'ERROR: {path} not found')
