#!/usr/bin/env python3
"""从搜索聚合页提取真实文章链接和封面图，下载到本地"""
import json, os, sys, re, urllib.request
from datetime import datetime
from urllib.parse import quote, urlparse, parse_qs, unquote
from playwright.sync_api import sync_playwright

DEEPSEEK_API_URL = "https://api.deepseek.com/v1/chat/completions"
CHROME_PATH = "/root/.cache/ms-playwright/chromium-1217/chrome-linux64/chrome"
_img_only_mode = False

def _get_deepseek_key():
    global DEEPSEEK_API_KEY
    try:
        with open("/root/.openclaw/openclaw.json") as f:
            cfg = json.load(f)
        return cfg['models']['providers']['deepseek']['apiKey']
    except:
        return ""

def download_img(img_url, i, item, img_dir, mmdd, source, key_name):
    fname = f"{key_name}_{i}.jpg"
    local_path = os.path.join(img_dir, fname)
    try:
        urllib.request.urlretrieve(img_url, local_path)
        sz = os.path.getsize(local_path)
        if sz > 1000:
            item['thumb'] = img_dir.rstrip('/').replace('/data/news', '') + '/' + fname
            print(f"    OK [{source}] {fname} ({sz//1024}K)")
            # 大于200K才压缩
            if sz > 200 * 1024:
                try:
                    import subprocess as _sp
                    _ext = local_path.lower()
                    sz_before = os.path.getsize(local_path)
                    if _ext.endswith('.png'):
                        _sp.run(["pngquant", "--quality=70-90", "--speed=1", "--force", "--ext", ".png", local_path],
                                capture_output=True, timeout=30)
                        _sp.run(["optipng", "-o2", local_path],
                                capture_output=True, timeout=30)
                    elif _ext.endswith('.jpg') or _ext.endswith('.jpeg'):
                        _tmp = local_path + ".tmp.jpg"
                        _sp.run(["cjpeg", "-quality", "85", "-outfile", _tmp, local_path],
                                capture_output=True, timeout=30)
                        if os.path.exists(_tmp) and os.path.getsize(_tmp) > 1024:
                            os.replace(_tmp, local_path)
                        elif os.path.exists(_tmp):
                            os.remove(_tmp)
                    sz_after = os.path.getsize(local_path)
                    if sz_after < sz_before:
                        saved = (sz_before - sz_after) * 100 // sz_before
                        print(f"      \u2139 \u538b\u7f29: {sz_before//1024}K \u2192 {sz_after//1024}K (-{saved}%)")
                except:
                    pass
        else:
            print(f"    WARN 图片太小({sz}bytes)")
            os.remove(local_path)
    except Exception as e:
        print(f"    FAIL 下载失败: {str(e)[:30]}")

def gen_svg_fallback(item, title, i, img_dir, mmdd, key_name):
    import textwrap
    lines = textwrap.wrap(title, width=12) if len(title) > 12 else [title]
    start_y = 150 - (len(lines) - 1) * 22
    txt = ""
    for idx, line in enumerate(lines):
        y = start_y + idx * 45
        txt += f'<text x="200" y="{y}" text-anchor="middle" fill="white" font-size="28" font-family="sans-serif">{line}</text>\n'
    svg = f'<svg xmlns="http://www.w3.org/2000/svg" width="400" height="300" viewBox="0 0 400 300">\n<rect width="400" height="300" fill="#1e40af" rx="12"/>\n{txt}</svg>'
    fname = key_name + "_" + str(i) + ".svg"
    with open(os.path.join(img_dir, fname), "w", encoding="utf-8") as f:
        f.write(svg)
    item["thumb"] = img_dir.replace('/data/news', '') + "/" + fname
    print("    ▶ [SVG] " + fname)

def gen_summary_and_focus(item, article_url, is_search_source):
    global _img_only_mode
    if _img_only_mode or not is_search_source:
        return
    if item.get('summary') and item.get('focus_point'):
        return
    # 微博搜索页（需登录）直接基于标题，不走AI
    if 's.weibo.com/weibo' in article_url:
        import re as _rr
        _t = item.get('title', '')
        _clean = _rr.sub(r'\s+\d+[\.\d]*\s*[万千亿]?\s*$', '', _t).strip()
        item['summary'] = _clean + '事件，相关话题在微博引发广泛关注和讨论。'
        item['focus_point'] = '借势微博热搜话题开展品牌情感营销，强化品牌人文关怀心智。'
        print(f"      ✅ 微博项目直接生成")
        return
    key = _get_deepseek_key()
    if not key:
        return
    try:
        req = urllib.request.Request(f"https://r.jina.ai/{article_url}", headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        text = resp.read().decode('utf-8')[:2000]
    except:
        text = ""

    prompt = f'阅读文章内容\"{text}\"或者标题\"{item.get("title","")}\"生成JSON生成JSON（仅JSON）：\n1. summary：不超过40汉字点出核心事件\n2. focus_point：从现代汽车品牌（油车为主的车企）营销角度进行分析，提出对现代品牌有实际价值的营销启示和建议，需紧扣现代品牌营销（如数字化转型、用户运营、场景营销、体验营销、本土化策略等），30-35个汉字。\n{{"summary": "","focus_point": ""}}'
    
    try:
        payload = json.dumps({"model": "deepseek-chat","messages": [{"role": "user","content": prompt}],"temperature": 0.3,"max_tokens": 500}).encode()
        req2 = urllib.request.Request(DEEPSEEK_API_URL, data=payload, headers={'Authorization': f'Bearer {key}','Content-Type': 'application/json'})
        reply = json.loads(urllib.request.urlopen(req2, timeout=30).read().decode('utf-8'))['choices'][0]['message']['content']
        try:
            p = json.loads(reply)
            if p.get('summary'): item['summary'] = p['summary'][:40]
            if p.get('focus_point'): item['focus_point'] = p['focus_point'][:35]
        except:
            s = re.search(r'"summary"[^:]*:"([^"]+)"', reply)
            f = re.search(r'"focus_point"[^:]*:"([^"]+)"', reply)
            if s: item['summary'] = s.group(1)[:40]
            if f: item['focus_point'] = f.group(1)[:35]
        print(f"    ✅ summary+focus_point")
    except:
        pass

def extract_search_source_url(page, url):
    """从搜索页提取第一条文章的真实链接和来源"""
    if 'so.toutiao.com/search' not in url and 's.weibo.com/weibo' not in url:
        return None, None
    try:
        _result = page.evaluate("""() => {
            const links = document.querySelectorAll('a');
            const sources = ['第一财经','新浪财经','澎湃新闻','央视新闻','新华社','北京日报','上观新闻','新华网','人民网','观察者网','36氪','虎嗅','财联社','中国基金报','证券时报','经济观察报','21世纪经济报道','每日经济新闻','界面新闻','中国新闻网','光明网'];
            for (const a of links) {
                const h = a.href || '';
                const t = a.textContent.trim();
                if (h.includes('search/jump') && h.includes('url=') && t.length > 5) {
                    let card = a.closest('[class*="card"], [class*="item"], li, div');
                    let cardText = card ? card.textContent : '';
                    let source = '';
                    for (const s of sources) {
                        if (cardText.includes(s)) { source = s; break; }
                    }
                    try {
                        const u = new URL(h).searchParams.get('url');
                        if (u) return { url: decodeURIComponent(u).split('?')[0], source: source };
                    } catch(e) {}
                }
            }
            return null;
        }""")
        if _result:
            return _result['url'], _result['source']
    except:
        pass
    return None, None

def proc_item(item, browser, img_dir, mmdd, i, key_name, img_only=False):
    global _img_only_mode
    _img_only_mode = img_only
    url = item.get('source_url', '')
    if not url: return
    title = item.get('title', '')
    title = title[:30]
    print(f"\n  [{i}] {title}")

    thumb_url = item.get('thumb', '')
    if thumb_url and (thumb_url.startswith('http://') or thumb_url.startswith('https://')):
        print(f"    ⬇ 下载已有thumb...")
        download_img(thumb_url, i, item, img_dir, mmdd, "直链", key_name)
        if item.get('thumb', '').startswith('/images/'):
            print(f"    ✅ 直链成功")
            return
        print(f"    ⚠ 直链失败，继续")

    page = browser.new_page()
    if "s.weibo.com" in url:
        print("    ⏯ 微博需登录，直接SVG")
        page.close()
        gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)
        return

    try:
        page.goto(url, wait_until="domcontentloaded", timeout=20000)
        page.wait_for_timeout(5000)

        # 从搜索页提取真实文章链接
        real_url, source_name = extract_search_source_url(page, url)
        if real_url:
            item['source_url'] = real_url
            print(f"      ⚠ 更新 source_url: {real_url[:60]}...")
            if source_name:
                item['platform'] = source_name
                print(f"      ⚠ 更新 platform: {source_name}")

        elif any(d in url for d in ['m.toutiao.com','m.toutiaoimg.cn','article.zlink','view.inews.qq.com','news.qq.com','news.cn','xinhuanet.com','36kr.com','thepaper.cn']):
            # 已经是文章页，直接提取标题和来源
            try:
                _art_title = page.evaluate("document.title")
                if _art_title and len(_art_title) > 5:
                    _clean = _art_title.rsplit(' - ', 1)[0].rsplit('_', 1)[0].strip()
                    if len(_clean) > 5:
                        item['title'] = _clean
                        print(f"      ⚠ 更新 title: {_clean[:40]}...")
                _pfs = ["第一财经","新浪财经","澎湃新闻","央视新闻","新华社","北京日报","上观新闻","新华网","人民网","观察者网","36氪","虎嗅","财联社","中国基金报","证券时报","经济观察报","21世纪经济报道","每日经济新闻","界面新闻","中国新闻网","光明网","中国经营报","国际金融报","中国证券报"]
                _pt = page.evaluate("document.body.innerText.substring(0,5000)") or ""
                _found = ""
                for _s in _pfs:
                    if _s in _pt:
                        _found = _s
                        break
                if not _found:
                    # 通过域名映射
                    _domain_map = {"view.inews.qq.com": "腾讯新闻","news.qq.com": "腾讯新闻","www.news.cn": "新华网","xinhuanet.com": "新华网","36kr.com": "36氪","thepaper.cn": "澎湃新闻","sohu.com": "搜狐","163.com": "网易"}
                    for _dm, _dp in _domain_map.items():
                        if _dm in (item.get('source_url','') or ''):
                            _found = _dp
                            break
                if _found:
                    item["platform"] = _found
                    print(f"      ⚠ 更新 platform: {_found}")
            except:
                pass

        img_url = page.evaluate("""() => {
            const imgs = document.querySelectorAll('img');
            for (const img of imgs) {
                const src = img.src || '';
                const w = img.naturalWidth || img.width || 0;
                const h = img.naturalHeight || img.height || 0;
                if (src && src.startsWith('http') && (w > 400 || h > 400)) {
                    const skip = ['avatar','logo','icon','favicon','300x300','emoji','sns_icon','browser_icon'];
                    if (!skip.some(s => src.includes(s))) return src;
                }
            }
            return '';
        }""")

        if img_url:
            download_img(img_url, i, item, img_dir, mmdd, "文章页", key_name)
            if not _img_only_mode:
                gen_summary_and_focus(item, item.get('source_url', url), True)
            page.close()
            return

        print("    ⚠ 文章页无图，直接SVG")
        page.close()
    except Exception as e:
        print(f"    ⚠ 文章页失败({str(e)[:30]})，直接SVG")
        page.close()

    is_search = 'so.toutiao.com' in url or 's.weibo.com' in url or 'm.toutiao.com' in url or 'm.toutiaoimg.cn' in url
    gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)
    gen_summary_and_focus(item, item.get('source_url', url), is_search)

def baidu_fallback(browser, title, i, item, img_dir, mmdd, key_name):
    """百度搜索（已禁用）直接SVG"""
    gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)

def main():
    import argparse
    parser = argparse.ArgumentParser(description="处理社会热点图片和摘要")
    parser.add_argument("input", help="JSON文件路径")
    parser.add_argument("key", help="字段名")
    parser.add_argument("img_dir", nargs="?", default="", help="图片目录")
    parser.add_argument("--img-only", action="store_true", help="仅下载图片")
    parser.add_argument("--summary-only", action="store_true", help="仅生成summary")
    args = parser.parse_args()
    if args.img_only and args.summary_only:
        print("❌ --img-only 和 --summary-only 不能同时使用"); sys.exit(1)
    if not args.img_dir and not args.summary_only:
        print("❌ 非summary-only模式时img_dir必填"); sys.exit(1)

    in_path = args.input; out_path = args.input
    keys = [k.strip() for k in args.key.split(",") if k.strip()]
    img_dir = args.img_dir
    with open(in_path, encoding='utf-8') as f: _raw_data = json.load(f)

    for json_key in keys:

    need_proc = []; need_summary = []
    print(f"[{datetime.now().strftime('%H:%M:%S')}] {json_key}")
        data = json.loads(json.dumps(_raw_data))  # 深拷贝
        if isinstance(data, dict): spots = data.get(json_key, [])
        elif isinstance(data, list):
            spots = []
            for item in data:
                if item.get('name') == json_key: spots = item.get('list', []); break
        else: print("❌ 未知结构"); sys.exit(1)
        print(f"  {json_key}: {len(spots)} 条")

        for i, item in enumerate(spots):
        url = item.get('source_url', '')
        thumb = item.get('thumb', '')
        if not url: continue
        if not thumb or thumb.startswith('http://') or thumb.startswith('https://') or 'article.zlink' in thumb:
            need_proc.append((i, item))
        if url and (not item.get('summary') or not item.get('focus_point')):
            need_summary.append((i, item))

    if args.summary_only:
        # 清洗所有标题
        for _si, _sitem in enumerate(spots):
            _ot = _sitem.get('title', '')
            _ct = re.sub(r'\s+\d+[\.\d]*\s*[万千亿]?\s*$', '', _ot).strip()
            if _ct and _ct != _ot:
                _sitem['title'] = _ct
        if not need_summary:
            print("  ✅ 无需处理summary")
        else:
            print(f"\n生成summary ({len(need_summary)}条)...")
            with sync_playwright() as pw:
                b = pw.chromium.launch(executable_path=CHROME_PATH, headless=True, args=["--no-sandbox"])
                for idx, (si, sitem) in enumerate(need_summary):
                    url = sitem.get('source_url', '')
                    if 'so.toutiao.com/search' in url or 's.weibo.com/weibo' in url:
                        try:
                            p = b.new_page()
                            p.goto(url, wait_until="domcontentloaded", timeout=20000)
                            p.wait_for_timeout(5000)
                            ru, sn = extract_search_source_url(p, url)
                            if ru:
                                sitem['source_url'] = ru
                                print(f"      ⚠ source_url updated: {ru[:60]}...")
                                if sn:
                                    sitem['platform'] = sn
                                    print(f"      ⚠ platform: {sn}")
                            p.close()
                        except: pass
                    gen_summary_and_focus(sitem, sitem.get('source_url', url), True)
                b.close()
        with open(out_path, 'w', encoding='utf-8') as f: json.dump(data, f, ensure_ascii=False, indent=2)
        print(f"\n✅ 保存到 {out_path}")
        return

    # 也处理已有thumb但source_url仍是搜索链接的项
    
    # --img-only 时，只处理无thumb或http外链，跳过已有本地路径的
    if not args.img_only:
        print(f"[{datetime.now().strftime('%H:%M:%S')}] {json_key}")
        data = json.loads(json.dumps(_raw_data))  # 深拷贝
        if isinstance(data, dict): spots = data.get(json_key, [])
        elif isinstance(data, list):
            spots = []
            for item in data:
                if item.get('name') == json_key: spots = item.get('list', []); break
        else: print("❌ 未知结构"); sys.exit(1)
        print(f"  {json_key}: {len(spots)} 条")

        for i, item in enumerate(spots):
            url = item.get('source_url', '')
            if url and (i, item) not in need_proc:
                need_proc.append((i, item))
    else:
        print(f"[{datetime.now().strftime('%H:%M:%S')}] {json_key}")
        data = json.loads(json.dumps(_raw_data))  # 深拷贝
        if isinstance(data, dict): spots = data.get(json_key, [])
        elif isinstance(data, list):
            spots = []
            for item in data:
                if item.get('name') == json_key: spots = item.get('list', []); break
        else: print("❌ 未知结构"); sys.exit(1)
        print(f"  {json_key}: {len(spots)} 条")

        for i, item in enumerate(spots):
            url = item.get('source_url', '')
            thumb = item.get('thumb', '')
            need_plat = url and not item.get('platform') and ('m.toutiao.com' in url or 'view.inews.qq.com' in url or 'news.qq.com' in url or 'news.cn' in url or 'www.news.cn' in url or 'xinhuanet.com' in url or '36kr.com' in url)
            if url and (i, item) not in need_proc and ((not thumb or thumb.startswith('http://') or thumb.startswith('https://')) or need_plat):
                need_proc.append((i, item))
    if not need_proc:
        print("  ✅ 无需处理"); return
    print(f"  需处理: {len(need_proc)} 条")

    mmdd = os.path.basename(img_dir)
    os.makedirs(img_dir, exist_ok=True)
    with sync_playwright() as pw:
        browser = pw.chromium.launch(executable_path=CHROME_PATH, headless=True, args=["--no-sandbox"])
        for idx, (i, item) in enumerate(need_proc):
            proc_item(item, browser, img_dir, mmdd, i, json_key, args.img_only)
        browser.close()

    if need_summary and not args.img_only:
        print(f"\n生成summary+focus ({len(need_summary)}条)...")
        for idx, (i, item) in enumerate(need_summary):
            gen_summary_and_focus(item, item.get('source_url', item.get('source_url','')), True)

    with open(out_path, 'w', encoding='utf-8') as f: json.dump(data, f, ensure_ascii=False, indent=2)
    print(f"\n✅ 保存到 {out_path}")

if __name__ == '__main__':
    main()
