#!/usr/bin/env python3
"""从搜索聚合页提取真实文章链接和封面图，下载到本地"""
import json, os, sys, re, urllib.request
from datetime import datetime
from urllib.parse import quote
from playwright.sync_api import sync_playwright

# DeepSeek API 配置
DEEPSEEK_API_KEY = None
DEEPSEEK_API_URL = "https://api.deepseek.com/v1/chat/completions"

def _get_deepseek_key():
    """延迟加载DeepSeek API key"""
    global DEEPSEEK_API_KEY
    if DEEPSEEK_API_KEY is None:
        try:
            with open("/root/.openclaw/openclaw.json") as f:
                cfg = json.load(f)
            DEEPSEEK_API_KEY = cfg['models']['providers']['deepseek']['apiKey']
        except:
            DEEPSEEK_API_KEY = ""
    return DEEPSEEK_API_KEY

def gen_summary_and_focus(item, article_url, is_search_source):
    """从文章内容生成summary和focus_point"""
    global _img_only_mode
    if _img_only_mode or not is_search_source:
        return
    if item.get('summary') and item.get('focus_point'):
        return
    try:
        # 0. 如果是搜索聚合页链接，先提取真实文章URL
        fetch_url = article_url
        if 'so.toutiao.com/search' in article_url or 's.weibo.com/weibo' in article_url:
            try:
                req0 = urllib.request.Request(f"https://r.jina.ai/{article_url}", headers={'User-Agent': 'Mozilla/5.0'})
                resp0 = urllib.request.urlopen(req0, timeout=15)
                text0 = resp0.read().decode('utf-8')
                # 从搜索页提取第一条m.toutiao.com/group/链接
                import re as _re
                links = []
                # 匹配markdown链接 [text](url) 中的真实文章URL
                for m in _re.finditer(r'https://m\.(?:toutiao|toutiaoimg)\.(?:com|cn)/group/\d+', text0):
                    links.append(m.group())
                if not links:
                    # 直接匹配URL
                    links = _re.findall(r'https://m\.(?:toutiao|toutiaoimg)\.(?:com|cn)/group/\d+', text0)
                if links:
                    fetch_url = links[0]
            except:
                pass
        # 1. 拉取文章内容
        req = urllib.request.Request(f"https://r.jina.ai/{fetch_url}", headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        text = resp.read().decode('utf-8')
        content = text[:2000]

        # 2. 调用DeepSeek生成
        key = _get_deepseek_key()
        if not key:
            return

        if len(content.strip()) < 100:
            prompt = f'''基于标题"{item.get("title","")}"生成JSON（仅JSON）：
1. summary：根据标题内容直接总结，不超过40个汉字
2. focus_point：结合标题事件，从现代汽车品牌营销角度给出具体可执行的启示建议，30-35个汉字，不能是泛泛的"加强营销"之类

{{"summary": "", "focus_point": ""}}'''
        else:
            prompt = f'''阅读以下文章内容，按要求生成JSON（仅输出JSON，不要任何其他文字）：

【要求】
1. summary：根据文章内容自己总结，不超过40个汉字，要包含核心事件和关键信息
2. focus_point：结合文章内容，从现代汽车品牌营销角度进行深度分析，提出对现代品牌有实际价值的营销启示和建议，需紧扣现代品牌营销（如数字化转型、用户运营、体验营销、本土化策略等），30-35个汉字。不能写"加强某方面"这种空话，要具体到活动形式、策略方向。

【文章内容】
{content[:2500]}

{{"summary": "", "focus_point": ""}}''' 

        payload = json.dumps({
            "model": "deepseek-chat",
            "messages": [{"role": "user", "content": prompt}],
            "temperature": 0.3,
            "max_tokens": 500
        }).encode()

        req2 = urllib.request.Request(DEEPSEEK_API_URL, data=payload,
            headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
        resp2 = urllib.request.urlopen(req2, timeout=30)
        result = json.loads(resp2.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content']

        # 3. 解析JSON
        try:
            parsed = json.loads(reply)
            if parsed.get('summary'):
                item['summary'] = parsed['summary'][:40]
            if parsed.get('focus_point'):
                item['focus_point'] = parsed['focus_point'][:35]
            print(f"    \u2705 summary+focus_point \u5df2\u751f\u6210")
            print(f"      summary: {item.get('summary','')[:30]}...")
            print(f"      focus: {item.get('focus_point','')[:30]}...")
        except:
            # 尝试正则提取
            import re as _re
            s = _re.search(r'"summary"[^:]*:"([^"]+)"', reply)
            f = _re.search(r'"focus_point"[^:]*:"([^"]+)"', reply)
            if s: item['summary'] = s.group(1)[:40]
            if f: item['focus_point'] = f.group(1)[:35]
    except:
        pass


ORIGIN_DIR = "/data/news/json/origin_data/"
CHROME_PATH = "/root/.cache/ms-playwright/chromium-1217/chrome-linux64/chrome"


def download_img(img_url, i, item, img_dir, mmdd, source, key_name):
    """下载图片到本地"""
    fname = f"{key_name}_{i}.jpg"
    local_path = os.path.join(img_dir, fname)
    try:
        urllib.request.urlretrieve(img_url, local_path)
        sz = os.path.getsize(local_path)
        if sz > 1000:
            item['thumb'] = f"/images/{mmdd}/{fname}"
            print(f"    OK [{source}] {fname} ({sz//1024}K)")
        else:
            print(f"    WARN 图片太小({sz}bytes)")
            os.remove(local_path)
    except Exception as e:
        print(f"    FAIL 下载失败: {str(e)[:30]}")

_img_only_mode = False

def proc_item(item, browser, img_dir, mmdd, i, key_name, img_only=False):
    global _img_only_mode
    _img_only_mode = img_only
    """处理单条social热点"""
    url = item.get('source_url', '')
    if not url:
        return

    title = item.get('title', '')[:30]
    print(f"\n  [{i}] {title}")

    # 如果已有http/https thumb，先尝试直接下载
    thumb_url = item.get('thumb', '')
    if thumb_url and (thumb_url.startswith('http://') or thumb_url.startswith('https://')):
        print(f"    ⬇️ 下载已有thumb...")
        download_img(thumb_url, i, item, img_dir, mmdd, "直链", key_name)
        # 检查是否下载成功（thumb被更新为本地路径）
        if item.get('thumb', '').startswith('/images/'):
            print(f"    ✅ 直链下载成功")
            return
        else:
            print(f"    ⚠️ 直链下载失败，继续文章页")
    page = browser.new_page()

    if "s.weibo.com" in url:
        print("    \u23ef 微博需登录，直接SVG兜底")
        page.close()
        gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)
        return

    try:
        page.goto(url, wait_until="domcontentloaded", timeout=20000)
        page.wait_for_timeout(5000)
        if "so.toutiao.com/search" in url or "s.weibo.com/weibo" in url:
            try:
                _j = page.evaluate("""() => {
                    const links = document.querySelectorAll("a");
                    for (const a of links) {
                        const h = a.href || "";
                        const t = a.textContent.trim();
                        if (h.includes("search/jump") && h.includes("url=") && t.length > 5) return h;
                    }
                    return "";
                }""")
                if _j:
                    from urllib.parse import urlparse, parse_qs, unquote
                    _u = parse_qs(urlparse(_j).query).get("url", [""])[0]
                    if _u:
                        item["source_url"] = unquote(_u).split("?")[0]
                        print(f"      \U000026a0 \u66f4\u65b0 source_url: {item['source_url'][:60]}...")
            except:
                pass

        img_url = page.evaluate("""() => {
            const imgs = document.querySelectorAll('img');
            for (const img of imgs) {
                const src = img.src || '';
                const w = img.naturalWidth || 0;
                if (src.includes('toutiaoimg') && !src.includes('avatar') && !src.includes('300x300') && w > 100) return src;
            }
            for (const img of imgs) {
                const src = img.src || '';
                if (src.includes('toutiaoimg') && !src.includes('avatar')) return src;
            }
            return '';
        }""")

        if img_url:
            download_img(img_url, i, item, img_dir, mmdd, "文章页", key_name)
            # 从搜索页提取真实source_url
            if 'so.toutiao.com/search' in url or 's.weibo.com/weibo' in url:
                try:
                    _jumplink = page.evaluate("""() => {
                        const links = document.querySelectorAll('a');
                        for (const a of links) {
                            const h = a.href || '';
                            if (h.includes('search/jump') && !h.includes('/c/user/')) return h;
                        }
                        return '';
                    }""")
                except:
                    pass
            # 从文章页内容生成summary
            if not _img_only_mode:

        print("    ⚠️ 文章页无图，直接SVG")
        page.close()
    except Exception as e:
        print(f"    ⚠️ 文章页失败({str(e)[:30]})，直接SVG")
        page.close()

    # 非搜索源不生成summary
    is_search = 'so.toutiao.com' in url or 's.weibo.com' in url or 'm.toutiao.com' in url or 'toutiaoimg' in url or 'article.zlink' in url

    print("    \u26a0 \u6587\u7ae0\u9875\u65e0\u56fe\uff0c\u76f4\u63a5SVG")
    gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)
    gen_summary_and_focus(item, item.get('source_url', ''), is_search)

def gen_svg_fallback(item, title, i, img_dir, mmdd, key_name):
    """生成蓝色背景SVG兜底图片"""
    import textwrap
    lines = textwrap.wrap(title, width=12) if len(title) > 12 else [title]
    total_lines = len(lines)
    start_y = 150 - (total_lines - 1) * 22
    txt = ""
    for idx, line in enumerate(lines):
        y = start_y + idx * 45
        txt += "<text x=\"200\" y=\"" + str(y) + "\" text-anchor=\"middle\" fill=\"white\" font-size=\"28\" font-family=\"sans-serif\">" + line + "</text>\n"
    svg = "<svg xmlns=\"http://www.w3.org/2000/svg\" width=\"400\" height=\"300\" viewBox=\"0 0 400 300\">\n"
    svg += "<rect width=\"400\" height=\"300\" fill=\"#1e40af\" rx=\"12\"/>\n"
    svg += txt
    svg += "</svg>"
    fname = key_name + "_" + str(i) + ".svg"
    local_path = os.path.join(img_dir, fname)
    with open(local_path, "w", encoding="utf-8") as f:
        f.write(svg)
    item["thumb"] = "/images/" + mmdd + "/" + fname
    print("    \342\226\266 [SVG] " + fname)


def baidu_fallback(browser, title, i, item, img_dir, mmdd, key_name):
    """百度图片搜索"""
    try:
        p2 = browser.new_page()
        kw = title[:20]
        bd_url = f"https://image.baidu.com/search/index?tn=baiduimage&word={quote(kw)}"
        p2.goto(bd_url, wait_until="domcontentloaded", timeout=30000)
        p2.wait_for_timeout(5000)
        p2.evaluate("window.scrollTo(0, document.body.scrollHeight/2)")
        p2.wait_for_timeout(2000)
        p2.evaluate("window.scrollTo(0, 0)")
        p2.wait_for_timeout(1000)
        bi = p2.evaluate("""() => {
            const imgs = document.querySelectorAll("img");
            for (const img of imgs) {
                const src = img.src || "";
                if (src.includes("img2.baidu.com") && src.includes("size=f") && src.length > 80) return src;
            }
            return "";
        }""")
        p2.close()
        if bi:
            bi = re.sub(r"size=f\d+,\d+", "size=f0,0", bi)
            download_img(bi, i, item, img_dir, mmdd, "百度", key_name)
            if not _img_only_mode:
                gen_summary_and_focus(item, item.get('source_url', ''), True)
        else:
            print("    \u26a0 \u767e\u5ea6\u65e0\u7ed3\u679c")
            gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)
    except Exception as e:
        print(f"    \u26a0 \u767e\u5ea6\u641c\u7d22\u5931\u8d25: {str(e)[:30]}")
        gen_svg_fallback(item, title, i, img_dir, mmdd, key_name)



def main():
    import argparse
    parser = argparse.ArgumentParser(description="从搜索页提取文章链接和封面图")
    parser.add_argument("input", help="JSON文件路径（必填）")
    parser.add_argument("key", help="JSON中要处理的字段名，如 social_hotspots（必填）")
    parser.add_argument("img_dir", nargs="?", default="", help="图片保存目录（--summary-only 时可省略）")
    parser.add_argument("--img-only", action="store_true", help="仅下载图片，不处理summary/focus_point等")
    parser.add_argument("--summary-only", action="store_true", help="仅生成summary和focus_point，不下载图片")
    args = parser.parse_args()
    if args.img_only and args.summary_only:
        print("❌ --img-only 和 --summary-only 不能同时使用")
        sys.exit(1)
    if not args.img_dir and not args.summary_only:
        print("❌ 非 --summary-only 模式时，img_dir 为必填参数")
        sys.exit(1)

    in_path = args.input
    out_path = args.input
    json_key = args.key
    img_dir = args.img_dir

    print(f"[{datetime.now().strftime('%H:%M:%S')}] 提取真实文章链接和封面图")
    print(f"  输入: {in_path}")

    with open(in_path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    if isinstance(data, dict):
        spots = data.get(json_key, [])
    elif isinstance(data, list):
        spots = []
        for item in data:
            if item.get('name') == json_key:
                spots = item.get('list', [])
                break
    else:
        print("❌ 未知数据结构")
        sys.exit(1)

    print(f"  {json_key}: {len(spots)} 条")

    # 找出需要处理的条目
    need_proc = []
    need_summary = []
    for i, item in enumerate(spots):
        url = item.get('source_url', '')
        thumb = item.get('thumb', '')
        if not url:
            continue
        # 需要处理的条件：无thumb / 外链thumb（防盗链） / http/https链接需下载到本地
        if not thumb or thumb.startswith('http://') or thumb.startswith('https://') or 'article.zlink' in thumb or '1x1.gif' in thumb:
            need_proc.append((i, item))
        # 搜索源条目需要summary/focus_point（即使已有本地thumb）
        if ('so.toutiao.com' in url or 's.weibo.com' in url or 'm.toutiao.com' in url or 'toutiaoimg' in url or 'article.zlink' in url) and (not item.get('summary') or not item.get('focus_point')):
            need_summary.append((i, item))

    if args.summary_only:
        if not need_summary:
            print("  ✅ 无需处理summary")
    if args.summary_only:
        if not need_summary:
            print("  ✅ 无需处理summary")
        else:
            print(f"\n=== 仅生成summary和focus_point ({len(need_summary)}条) ===")
            try:
                from playwright.sync_api import sync_playwright as _pw
                with _pw() as _pw_ctx:
                    _browser = _pw_ctx.chromium.launch(executable_path=CHROME_PATH, headless=True, args=["--no-sandbox"])
                    for idx, (si, sitem) in enumerate(need_summary):
                        url = sitem.get('source_url', '')
                        if 'so.toutiao.com/search' in url or 's.weibo.com/weibo' in url:
                            try:
                                _p = _browser.new_page()
                                _p.goto(url, wait_until="domcontentloaded", timeout=20000)
                                _p.wait_for_timeout(3000)
                                _real = _p.evaluate("""() => {
                                    const links = document.querySelectorAll("a");
                                    for (const a of links) {
                                        const h = a.href || "";
                                        if (h.includes("m.toutiao.com/group/") || h.includes("toutiaoimg.cn/group/")) return h;
                                    }
                                    for (const a of links) {
                                        const h = a.href || "";
                                        if (h.includes("toutiao.com/group/") && !h.includes("search/jump")) return h;
                                    }
                                    return "";
                                }""")
                                if _real:
                                    _real = _real.split("?")[0] if "?" in _real else _real
                                    sitem["source_url"] = _real
                                    print(f"      \u26a0 \u66f4\u65b0 source_url: {_real[:60]}...")
                                _p.close()
                            except:
                                pass
                        gen_summary_and_focus(sitem, sitem.get("source_url", ""), True)
                    _browser.close()
            except:
                pass
        with open(out_path, 'w', encoding='utf-8') as f:
            json.dump(data, f, ensure_ascii=False, indent=2)
        print(f"\n✅ 已保存到 {out_path}")
        return
        return

    print(f"  需处理: {len(need_proc)} 条")

    mmdd = os.path.basename(img_dir)
    os.makedirs(img_dir, exist_ok=True)
    print(f"  图片目录: {img_dir}")

    with sync_playwright() as pw:
        browser = pw.chromium.launch(executable_path=CHROME_PATH, headless=True, args=["--no-sandbox"])
        for idx, (i, item) in enumerate(need_proc):
            proc_item(item, browser, img_dir, mmdd, i, json_key, args.img_only)
        browser.close()

    # 处理summary/focus_point（--img-only 时不处理）
    if need_summary and not args.img_only:
        print(f"\n生成summary和focus_point ({len(need_summary)}条)...")
        for idx, (i, item) in enumerate(need_summary):
            url = item.get('source_url', '')
            gen_summary_and_focus(item, url, True)

    with open(out_path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print(f"\n✅ 已保存到 {out_path}")


if __name__ == '__main__':
    main()
