#!/usr/bin/env python3
"""从搜索聚合页提取真实文章链接和封面图，下载到本地"""
import json, os, sys, re, urllib.request
from datetime import datetime
from urllib.parse import quote
from playwright.sync_api import sync_playwright

# Doubao 图片生成 API 配置
DOUBAO_API_KEY = "ark-563a4210-d804-47fa-a37a-40962192119b-ce6db"
DOUBAO_API_URL = "https://ark.cn-beijing.volces.com/api/v3/images/generations"
DOUBAO_MODEL = "doubao-seedream-4-0-250828"

# DeepSeek API 配置
DEEPSEEK_API_KEY = None  # 2026-08-29: 改为动态读取，勿硬编码（旧key已失效401）
DEEPSEEK_API_URL = "https://api.deepseek.com/v1/chat/completions"

def _get_deepseek_key():
    """延迟加载DeepSeek API key（优先 ~/.hermes/config.yaml，兜底 openclaw.json）"""
    global DEEPSEEK_API_KEY
    if DEEPSEEK_API_KEY is None:
        import yaml
        candidates = [
            ("/home/ubuntu/.hermes/config.yaml", lambda c: c['providers']['deepseek']['api_key']),
            ("/root/.openclaw/openclaw.json", lambda c: c['models']['providers']['deepseek']['apiKey']),
        ]
        for path, getter in candidates:
            try:
                with open(path) as f:
                    if path.endswith('.yaml'):
                        cfg = yaml.safe_load(f)
                    else:
                        cfg = json.load(f)
                key = getter(cfg)
                if key:
                    DEEPSEEK_API_KEY = key
                    break
            except Exception:
                continue
        if not DEEPSEEK_API_KEY:
            DEEPSEEK_API_KEY = ""
    return DEEPSEEK_API_KEY

# 2026-09-16 用户新增：learn_records 已标注 focus_point 的条目标题集合
# --summary-only 模式下，这些条目的 focus_point 视为人工权威标注 → 跳过生成与截断
_LEARN_FOCUS_TITLES = None


def _load_learn_focus_titles(in_path):
    """从 learn_records 加载「已有 focus_point」的条目标题集合。

    文件选择规则与流水线一致：按 mtime 取最新
    （勿用 sort -t_ -k2，多版本 _1/_1_2 会误选旧版）。
    找不到文件或读取失败 → 返回空集合，不影响原有行为。
    """
    global _LEARN_FOCUS_TITLES
    if _LEARN_FOCUS_TITLES is not None:
        return _LEARN_FOCUS_TITLES
    _LEARN_FOCUS_TITLES = set()
    import glob
    m = re.search(r"(\d{4})data", os.path.basename(in_path or ""))
    if not m:
        return _LEARN_FOCUS_TITLES
    mmdd = m.group(1)
    cands = []
    for pat in ("/data/news/learn_records/yes_%sdata_*.json" % mmdd,
                "/data/news/learn_records/%sdata_*.json" % mmdd,
                "/data/news/learn_records/%sdata.json" % mmdd):
        cands += glob.glob(pat)
    if not cands:
        print("  \U0001F4DA learn_records 未找到 %s 标注文件，focus_point 正常生成" % mmdd)
        return _LEARN_FOCUS_TITLES
    latest = max(cands, key=os.path.getmtime)
    try:
        with open(latest, encoding="utf-8") as _lf:
            lr = json.load(_lf)
    except Exception as e:
        print("  \u26A0\uFE0F learn_records 读取失败(%s): %s" % (os.path.basename(latest), str(e)[:40]))
        return _LEARN_FOCUS_TITLES
    for sec in (lr if isinstance(lr, list) else [lr]):
        if not isinstance(sec, dict):
            continue
        for it in (sec.get("list") or []):
            if (it.get("focus_point") or "").strip():
                _LEARN_FOCUS_TITLES.add((it.get("title") or "").strip())
    print("  \U0001F4DA learn_records 来源: %s，已有 focus_point %d 条 → 跳过生成"
          % (os.path.basename(latest), len(_LEARN_FOCUS_TITLES)))
    return _LEARN_FOCUS_TITLES


def _keep_learn_focus(item):
    """该条目的 focus_point 是否来自 learn_records 人工标注（是则跳过生成/截断）"""
    if not _LEARN_FOCUS_TITLES:
        return False
    return (item.get("title") or "").strip() in _LEARN_FOCUS_TITLES


def gen_summary_and_focus(item, article_url, is_search_source):
    """从文章内容生成summary和focus_point"""
    global _img_only_mode
    if _img_only_mode or not is_search_source:
        return
    # 不管是否有summary，超长标题都要缩短
    t = item.get("title", "")
    if len(t) > 45:
        item["title"] = t[:34] + "…"
    # 非今日或为空的 publish_time 纠正为今日；始终截断时间点
    pt = item.get("publish_time", "")
    today_str = datetime.now().strftime("%Y-%m-%d")
    if not pt or pt < today_str:
        item["publish_time"] = today_str
    if len(item.get("publish_time", "")) > 10:
        item["publish_time"] = item["publish_time"][:10]
    if item.get("summary") and item.get("focus_point"):
        return
    try:
        fetch_url = article_url  # 默认直接用source_url
        if 'so.toutiao.com/search' in article_url or 's.weibo.com/weibo' in article_url:
            try:
                req0 = urllib.request.Request(f"https://r.jina.ai/{article_url}", headers={'User-Agent': 'Mozilla/5.0'})
                resp0 = urllib.request.urlopen(req0, timeout=15)
                text0 = resp0.read().decode('utf-8')
                # 从搜索页提取第一条m.toutiao.com/group/链接
                import re as _re
                links = []
                # 匹配markdown链接 [text](url) 中的真实文章URL
                for m in _re.finditer(r'https://m\.(?:toutiao|toutiaoimg)\.(?:com|cn)/group/\d+', text0):
                    links.append(m.group())
                if not links:
                    # 直接匹配URL
                    links = _re.findall(r'https://m\.(?:toutiao|toutiaoimg)\.(?:com|cn)/group/\d+', text0)
                if links:
                    fetch_url = links[0]
            except:
                pass
        # 1. 拉取文章内容
        req = urllib.request.Request(f"https://r.jina.ai/{fetch_url}", headers={'User-Agent': 'Mozilla/5.0'})
        resp = urllib.request.urlopen(req, timeout=15)
        text = resp.read().decode('utf-8')
        content = text[:2000]
        
        # 2. 从页面内容提取真实发布时间
        import re as _re_dt
        _real_date = '-'
        _head = text[:5000]
        for _dt_p in [r'article:published_time["\s>]+content="([^"]+)"', r'pubdate[=:]\s*["\']?(\d{4}-\d{2}-\d{2})',
                      r'datetime="(\d{4}-\d{2}-\d{2})', r'data-time="(\d{4}-\d{2}-\d{2})',
                      r'"datePublished"\s*:\s*"(\d{4}-\d{2}-\d{2})', r'"pubDate"\s*:\s*"(\d{4}-\d{2}-\d{2})',
                      r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})',
                      r'(\d{4}-\d{2}-\d{2})\s*[·]', r'(\d{4}-\d{2}-\d{2})[T ]\d{2}:\d{2}']:
            _m = _re_dt.search(_dt_p, _head)
            if _m:
                _real_date = _m.group(1)[:10]
                break
        if _real_date != '-' and _real_date != item.get('publish_time', '')[:10]:
            item['publish_time'] = _real_date
            print(f"    📅 发布时间: {_real_date}")
        item['publish_time_verified'] = _real_date
        
        # 反爬/登录页面检测：如内容含登录/验证等关键词，改用DeepSeek直接访问URL
        login_patterns = ['微博登录', '滑块验证', '安全验证', '环境异常', '扫码登录', '请完成验证', '找不到', '无法加载', '未提供', '无法访问', '访问被拒绝', '403错误', '无法获取']
        if any(p in content for p in login_patterns):
            print(f"    ⚠️ 检测到登录/反爬页面，改用DeepSeek直接访问URL")
            try:
                ds_url_req = urllib.request.Request(DEEPSEEK_API_URL,
                    data=json.dumps({"model":"deepseek-chat","messages":[{"role":"user",
                        "content":f"访问以下URL并根据文章内容生成30-40字的中文摘要，保留核心事实。直接输出摘要不要解释：\n\n{article_url}\n\n标题：{item.get('title','')}"}],
                        "temperature":0.3,"max_tokens":200}).encode(),
                    headers={'Authorization': f'Bearer {_get_deepseek_key()}', 'Content-Type': 'application/json'})
                ds_url_resp = urllib.request.urlopen(ds_url_req, timeout=30)
                ds_summary = json.loads(ds_url_resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
                if ds_summary:
                    item['summary'] = ds_summary[:40]
                    if not item['summary'].endswith('。'):
                        item['summary'] += '。'
                    print(f"    ✅ DeepSeek URL访问成功")
                    # 跳过后面的逻辑
                    raise Exception("login_page_fallback")
            except Exception as e:
                if str(e) != "login_page_fallback":
                    content = ''  # 清空，让后面的逻辑用标题生成

        # 2. 用DeepSeek生成summary
        key = _get_deepseek_key()
        if not key:
            return

        if len(content.strip()) < 100:
            sum_prompt = f'''基于标题"{item.get("title","")}"生成summary（仅输出JSON）：
只输出一个字段：summary，不超过40个汉字，客观概括核心事件。
{{"summary": ""}}'''
        else:
            sum_prompt = f'''阅读以下文章内容，生成summary（仅输出JSON）：
只输出一个字段：summary，不超过40个汉字，包含核心事件和关键信息。

【文章内容】
{content[:2500]}

{{"summary": ""}}'''

        payload = json.dumps({
            "model": "deepseek-chat",
            "messages": [{"role": "user", "content": sum_prompt}],
            "temperature": 0.3,
            "max_tokens": 200
        }).encode()

        req2 = urllib.request.Request(DEEPSEEK_API_URL, data=payload,
            headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
        resp2 = urllib.request.urlopen(req2, timeout=30)
        result = json.loads(resp2.read().decode('utf-8'))
        reply = result['choices'][0]['message']['content']

        try:
            parsed = json.loads(reply)
            if parsed.get('summary'):
                item['summary'] = parsed['summary']
        except:
            import re as _re
            m = _re.search(r'"summary"\s*:\s*"([^"]+)"', reply)
            if m:
                item['summary'] = m.group(1)

        # 3. 用Doubao生成focus_point（基于title+summary）
        # 2026-09-16 用户新增：learn_records 已标注 focus_point 的条目跳过生成
        keep_fp = _keep_learn_focus(item)
        fp_title = item.get('title', '')
        fp_summary = item.get('summary', '')
        fp_prompt = '根据以下新闻，为现代汽车写一条策略启示（focus_point）。要求：不超过35个汉字，措辞柔和，用"可借鉴""可参考""不妨""建议"等友好语气，避免生硬的"应"。只输出JSON：{"focus_point": ""}\n\n标题：' + fp_title + '\n摘要：' + (fp_summary or fp_title)

        fp_payload = json.dumps({
            "model": "deepseek-chat",
            "messages": [{"role": "user", "content": fp_prompt}],
            "temperature": 0.3,
            "max_tokens": 200
        }).encode()

        try:
            if keep_fp:
                print(f"    \u23ED\uFE0F learn_records 已有 focus_point，跳过生成")
            else:
                fp_req = urllib.request.Request(DEEPSEEK_API_URL, data=fp_payload,
                    headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
                fp_resp = urllib.request.urlopen(fp_req, timeout=30)
                fp_result = json.loads(fp_resp.read().decode('utf-8'))
                fp_reply = fp_result['choices'][0]['message']['content']

                try:
                    fp_parsed = json.loads(fp_reply)
                    if fp_parsed.get('focus_point'):
                        item['focus_point'] = fp_parsed['focus_point']
                except:
                    import re as _re
                    m = _re.search(r'"focus_point"\s*:\s*"([^"]+)"', fp_reply)
                    if m:
                        item['focus_point'] = m.group(1)

            # 长度校验：summary≤40字，超长则用DeepSeek重生成；必须以。结尾
            s = item.get('summary', '')
            
            # 反爬提示检测：如摘要含环境异常/验证/滑块等关键词，用标题重生成
            anti_patterns = ['环境异常', '完成验证', '验证才能继续', '微信公众平台', '滑块验证', '安全验证', '扫码', '登录', '无法访问', '找不到', '无法加载', '无有效信息', '未提供', '访问被拒绝', '403错误', '无法获取']
            if any(p in s for p in anti_patterns):
                print(f"    ⚠️ 反爬提示检测到，用标题重生成")
                try:
                    title = item.get('title', '')
                    regenerate_req = urllib.request.Request(DEEPSEEK_API_URL,
                        data=json.dumps({"model":"deepseek-chat","messages":[{"role":"user",
                            "content":f"根据新闻标题\"{title}\"生成20-40字摘要，客观概括核心事件。直接输出摘要不要解释。"}],
                            "temperature":0.3,"max_tokens":100}).encode(),
                        headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
                    regenerate_resp = urllib.request.urlopen(regenerate_req, timeout=15)
                    new_s = json.loads(regenerate_resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
                    if new_s and not any(p in new_s for p in anti_patterns):
                        item['summary'] = new_s[:40]
                        s = item['summary']
                        print(f"    ✅ 根据标题重生成({len(s)}字): {s[:30]}...")
                except:
                    pass
                if not s:  # 重生成失败时清空
                    item.pop('summary', None)
                    s = ''
            
            if len(s) > 40:
                try:
                    shorten_req = urllib.request.Request(DEEPSEEK_API_URL,
                        data=json.dumps({
                            "model": "deepseek-chat",
                            "messages": [{"role": "user",
                                "content": f"将以下摘要压缩到40字以内，保留核心事件和关键信息，不要解释，不要省略号：\n\n{s}"}],
                            "temperature": 0.3,
                            "max_tokens": 100
                        }).encode(),
                        headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
                    shorten_resp = urllib.request.urlopen(shorten_req, timeout=15)
                    shorten_reply = json.loads(shorten_resp.read().decode('utf-8'))['choices'][0]['message']['content'].strip()
                    item['summary'] = shorten_reply[:40]
                    print(f"    ✅ DeepSeek重生成({len(item['summary'])}字)")
                except:
                    item['summary'] = s[:40]
                    print(f"    ⚠️ summary超长({len(s)}字)，截断至40字")
            if item.get('summary', '') and not item['summary'].endswith('。'):
                item['summary'] += '。'
            f = item.get('focus_point', '')
            if len(f) > 35 and not keep_fp:
                item['focus_point'] = f[:33] + '…'
                print(f"    ⚠️ focus_point超长({len(f)}字)，已截断")

            print(f"    ✅ summary({len(item.get('summary',''))}字) + focus_point({len(item.get('focus_point',''))}字)")
        except Exception as e:
            print(f"    ⚠️ focus_point生成失败: {str(e)[:40]}")
            # 降级：用DeepSeek再试一次
            try:
                fp_req2 = urllib.request.Request(DEEPSEEK_API_URL, data=payload2,
                    headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
                fp_resp2 = urllib.request.urlopen(fp_req2, timeout=30)
                fp_result2 = json.loads(fp_resp2.read().decode('utf-8'))
                fp_reply2 = fp_result2['choices'][0]['message']['content']
                m = re.search(r'"focus_point"\s*:\s*"([^"]+)"', fp_reply2)
                if m and not keep_fp:
                    item['focus_point'] = m.group(1)
                    print(f"    ✅ focus_point降级成功")
            except:
                pass
            s = _re.search(r'"summary"[^:]*:"([^"]+)"', reply)
            f = _re.search(r'"focus_point"[^:]*:"([^"]+)"', reply)
            if s: item['summary'] = s.group(1)[:40]
            if f and not keep_fp: item['focus_point'] = f.group(1)[:35]

        # 标题过长时自动缩短（超过45字→35字以内）
        title = item.get('title', '')
        if len(title) > 45:
            try:
                shorten_prompt = f'''将以下标题缩短到35字以内，保留核心信息，不要解释：

{title}'''
                sp = json.dumps({
                    "model": "deepseek-chat",
                    "messages": [{"role": "user", "content": shorten_prompt}],
                    "temperature": 0.1,
                    "max_tokens": 100
                }).encode()
                sr = urllib.request.Request(DEEPSEEK_API_URL, data=sp,
                    headers={'Authorization': f'Bearer {key}', 'Content-Type': 'application/json'})
                srp = urllib.request.urlopen(sr, timeout=15)
                s_result = json.loads(srp.read().decode('utf-8'))
                short_title = s_result['choices'][0]['message']['content'].strip()
                if short_title and 10 < len(short_title) <= 35:
                    item['title'] = short_title
                    print(f"      ✂️ 标题缩短: {len(title)}字→{len(short_title)}字")
            except:
                # 降级：直接截断
                item['title'] = title[:34] + '…'
                print(f"      ✂️ 标题截断: {len(title)}字→35字")
    except:
        pass


ORIGIN_DIR = "/data/news/json/origin_data/"
CHROME_PATH = "/home/ubuntu/.cache/ms-playwright/chromium-1217/chrome-linux64/chrome"


def compress_image(local_path):
    """压缩图片：JPEG用jpegoptim，PNG用pngquant+optipng"""
    import subprocess
    if not os.path.exists(local_path):
        return
    before = os.path.getsize(local_path)
    if before < 1000:
        return
    # 检查文件类型（magic bytes）
    is_jpeg = False
    is_png = False
    try:
        with open(local_path, 'rb') as _f:
            _h = _f.read(8)
        if _h[:2] == b'\xff\xd8':
            is_jpeg = True
        elif _h[:8] == b'\x89PNG\r\n\x1a\n':
            is_png = True
    except:
        return

    try:
        if is_jpeg:
            subprocess.run(['jpegoptim', '--max=85', '--strip-all', '--all-progressive', local_path],
                           capture_output=True, timeout=30)
        elif is_png:
            subprocess.run(
                ['pngquant', '--quality=70-90', '--speed=1', '--force', '--ext', '.png', local_path],
                capture_output=True, timeout=30
            )
            png_path = local_path.rsplit('.', 1)[0] + '.png'
            if os.path.exists(png_path):
                subprocess.run(['optipng', '-o2', png_path], capture_output=True, timeout=30)
                after = os.path.getsize(png_path)
                if after < before:
                    os.replace(png_path, local_path)
                else:
                    os.remove(png_path)
    except Exception:
        pass
    finally:
        if os.path.exists(local_path):
            after_final = os.path.getsize(local_path)
            saved = before - after_final
            if saved > 1024:
                print(f"      📦 压缩: {before//1024}K → {after_final//1024}K (省{saved//1024}K)")


def download_img(img_url, i, item, img_dir, mmdd, source, key_name):
    """下载图片到本地"""
    fname = f"{key_name}_{i}.jpg"
    local_path = os.path.join(img_dir, fname)
    try:
        if 'ithome.com' in img_url or 'inews.gtimg.com' in img_url or 'gtimg.com' in img_url:
            req = urllib.request.Request(img_url, headers={'Referer': 'https://www.ithome.com/', 'User-Agent': 'Mozilla/5.0'})
            with urllib.request.urlopen(req, timeout=15) as r:
                with open(local_path, 'wb') as f:
                    f.write(r.read())
        else:
            # ⚠️ urlretrieve 无 timeout，慢源(如 adquan/oss)会无限挂起（2026-09-02 修复）
            req = urllib.request.Request(img_url, headers={'User-Agent': 'Mozilla/5.0'})
            with urllib.request.urlopen(req, timeout=20) as r:
                with open(local_path, 'wb') as f:
                    f.write(r.read())
        sz = os.path.getsize(local_path)
        if sz > 1000:
            item['thumb'] = f"/images/{mmdd}/{fname}"
            print(f"    OK [{source}] {fname} ({sz//1024}K)")
            compress_image(local_path)
        else:
            print(f"    WARN 图片太小({sz}bytes)")
            os.remove(local_path)
    except Exception as e:
        print(f"    FAIL 下载失败: {str(e)[:30]}")

_img_only_mode = False

def proc_item(item, browser, img_dir, mmdd, i, key_name, img_only=False):
    global _img_only_mode
    _img_only_mode = img_only
    """处理单条social热点"""
    url = item.get('source_url', '')
    if not url:
        return

    title = item.get('title', '')[:30]
    print(f"\n  [{i}] {title}")

    # 如果已有http/https thumb，先尝试直接下载
    thumb_url = item.get('thumb', '')
    if thumb_url and (thumb_url.startswith('http://') or thumb_url.startswith('https://')):
        print(f"    ⬇️ 下载已有thumb...")
        download_img(thumb_url, i, item, img_dir, mmdd, "直链", key_name)
        # 检查是否下载成功（thumb被更新为本地路径）
        if item.get('thumb', '').startswith('/images/'):
            print(f"    ✅ 直链下载成功")
            return
        else:
            print(f"    ⚠️ 直链下载失败，继续文章页")
    page = browser.new_page()

    if "s.weibo.com" in url:
        print("    \u23ef 微博需登录，直接SVG兜底")
        page.close()
        gen_doubao_image(item, title, i, img_dir, mmdd, key_name, item.get("summary",""))
        return

    try:
        page.goto(url, wait_until="domcontentloaded", timeout=20000)
        page.wait_for_timeout(3000)
        # 滚动触发懒加载
        # 多步滚动触发懒加载
        for _s_i in range(6):
            page.evaluate(f"window.scrollTo(0, document.body.scrollHeight * {(_s_i+1)} / 6)")
            page.wait_for_timeout(500)
        page.evaluate("window.scrollTo(0, 0)")
        page.wait_for_timeout(1000)
        if "so.toutiao.com/search" in url or "s.weibo.com/weibo" in url:
            try:
                _j = page.evaluate("""() => {
                    const links = document.querySelectorAll("a");
                    for (const a of links) {
                        const h = a.href || "";
                        const t = a.textContent.trim();
                        if (h.includes("search/jump") && h.includes("url=") && t.length > 5) return h;
                    }
                    return "";
                }""")
                if _j:
                    from urllib.parse import urlparse, parse_qs, unquote
                    _u = parse_qs(urlparse(_j).query).get("url", [""])[0]
                    if _u:
                        item["source_url"] = unquote(_u).split("?")[0]
                        print(f"      \U000026a0 \u66f4\u65b0 source_url: {item['source_url'][:60]}...")
            except:
                pass

        img_url = page.evaluate("""() => {
            const imgs = document.querySelectorAll('img');
            let best = '';
            let bestScore = 0;
            for (const img of imgs) {
                const src = img.src || '';
                const w = img.naturalWidth || img.width || 0;
                const h = img.naturalHeight || img.height || 0;
                if (!src || !src.startsWith('http') || w < 150 || h < 150) continue;
                const skip = ['avatar','logo','icon','favicon','300x300','emoji','sns_icon','browser_icon'];
                if (skip.some(s => src.includes(s))) continue;
                const ratio = w / h;
                let score = 0;
                if (ratio >= 1.2 && ratio <= 2.0) score = 100 - Math.abs(ratio - 1.78) * 30;
                else if (ratio > 2.0) score = 50;
                else score = 10;
                score += Math.min(w * h / 10000, 50);
                if (src.includes('thumbnail') || src.includes('_240.') || src.includes('_200.')) score -= 40;
                if (score > bestScore) { bestScore = score; best = src; }
            }
            return best;
        }""")

        if img_url:
            download_img(img_url, i, item, img_dir, mmdd, "文章页", key_name)
            # 从搜索页提取真实source_url
            if 'so.toutiao.com/search' in url or 's.weibo.com/weibo' in url:
                try:
                    _jumplink = page.evaluate("""() => {
                        const links = document.querySelectorAll('a');
                        for (const a of links) {
                            const h = a.href || '';
                            if (h.includes('search/jump') && !h.includes('/c/user/')) return h;
                        }
                        return '';
                    }""")
                except:
                    pass
            # 从文章页内容生成summary
            if not _img_only_mode:
                gen_summary_and_focus(item, item.get("source_url", url), True)
            page.close()
            return
        print("    ⚠️ 文章页无图，直接SVG")
        page.close()
    except Exception as e:
        print(f"    ⚠️ 文章页失败({str(e)[:30]})，直接SVG")
        page.close()

    # 非搜索源不生成summary
    is_search = 'so.toutiao.com' in url or 's.weibo.com' in url or 'm.toutiao.com' in url or 'toutiaoimg' in url or 'article.zlink' in url

    print("    \u26a0 \u6587\u7ae0\u9875\u65e0\u56fe\uff0c\u76f4\u63a5SVG")
    gen_doubao_image(item, title, i, img_dir, mmdd, key_name, item.get("summary",""))
    gen_summary_and_focus(item, item.get('source_url', ''), is_search)

def gen_doubao_image(item, title, i, img_dir, mmdd, key_name, summary=''):
    """用豆包生成配图，替代SVG兜底"""
    # 反爬提示过滤：摘要含反爬内容时不加入提示词
    anti_patterns = ['环境异常', '完成验证', '验证才能继续', '微信公众平台', '滑块验证', '安全验证', '无法访问', '找不到', '无法加载', '未提供', '访问被拒绝']
    clean_summary = ''
    if summary and not any(p in summary for p in anti_patterns):
        clean_summary = summary
    
    # 清除标题末尾的热度/排名数据（如 "10.0万"、"163.8万"）
    clean_title = re.sub(r'\s+\d+\.?\d*\s*万\s*$', '', title).strip()
    if not clean_title:
        clean_title = title

    # 豆包配图规则（2026-09-02 用户新增）：提示词不体现人物，只关注事件与品牌
    NO_PERSON_SUFFIX = '，画面不出现任何人物姓名与肖像，只呈现事件场景和品牌产品元素'

    prompt = "配图：" + clean_title + "，新闻配图风格，写实，横向构图，16:9，高清" + NO_PERSON_SUFFIX
    if clean_summary:
        prompt = "配图：" + clean_title + "，" + clean_summary[:60] + "，新闻配图风格，写实，横向构图，16:9，高清" + NO_PERSON_SUFFIX

    payload = json.dumps({
        "model": DOUBAO_MODEL,
        "prompt": prompt,
        "size": "1280x720",
        "n": 1,
        "response_format": "url"
    }).encode('utf-8')

    try:
        req = urllib.request.Request(
            DOUBAO_API_URL, data=payload,
            headers={'Authorization': f'Bearer {DOUBAO_API_KEY}', 'Content-Type': 'application/json'},
            method='POST'
        )
        resp = urllib.request.urlopen(req, timeout=120)
        result = json.loads(resp.read().decode('utf-8'))
        img_url = result.get('data', [{}])[0].get('url', '')

        if not img_url:
            print(f'    ❌ 豆包未返回图片URL')
            return False

        fname = f"{key_name}_{i}.jpg"
        local_path = os.path.join(img_dir, fname)
        os.makedirs(img_dir, exist_ok=True)

        img_req = urllib.request.Request(img_url, headers={'User-Agent': 'Mozilla/5.0'})
        img_resp = urllib.request.urlopen(img_req, timeout=60)
        with open(local_path, 'wb') as f:
            f.write(img_resp.read())

        item['thumb'] = '/images/' + mmdd + '/' + fname
        size_kb = os.path.getsize(local_path) // 1024
        print(f'    🎨 [豆包] {fname} ({size_kb}KB)')
        return True

    except Exception as e:
        print(f'    ❌ 豆包生成失败: {str(e)[:50]}')
        return False


def baidu_fallback(browser, title, i, item, img_dir, mmdd, key_name):
    """百度图片搜索"""
    try:
        p2 = browser.new_page()
        kw = title[:20]
        bd_url = f"https://image.baidu.com/search/index?tn=baiduimage&word={quote(kw)}"
        p2.goto(bd_url, wait_until="domcontentloaded", timeout=30000)
        p2.wait_for_timeout(5000)
        p2.evaluate("window.scrollTo(0, document.body.scrollHeight/2)")
        p2.wait_for_timeout(2000)
        p2.evaluate("window.scrollTo(0, 0)")
        p2.wait_for_timeout(1000)
        bi = p2.evaluate("""() => {
            const imgs = document.querySelectorAll("img");
            for (const img of imgs) {
                const src = img.src || "";
                if (src.includes("img2.baidu.com") && src.includes("size=f") && src.length > 80) return src;
            }
            return "";
        }""")
        p2.close()
        if bi:
            bi = re.sub(r"size=f\d+,\d+", "size=f0,0", bi)
            download_img(bi, i, item, img_dir, mmdd, "百度", key_name)
            if not _img_only_mode:
                gen_summary_and_focus(item, item.get('source_url', ''), True)
        else:
            print("    \u26a0 \u767e\u5ea6\u65e0\u7ed3\u679c")
            gen_doubao_image(item, title, i, img_dir, mmdd, key_name, item.get("summary",""))
    except Exception as e:
        print(f"    \u26a0 \u767e\u5ea6\u641c\u7d22\u5931\u8d25: {str(e)[:30]}")
        gen_doubao_image(item, title, i, img_dir, mmdd, key_name, item.get("summary",""))



def main():
    import argparse
    parser = argparse.ArgumentParser(description="从搜索页提取文章链接和封面图")
    parser.add_argument("input", help="JSON文件路径（必填）")
    parser.add_argument("key", help="JSON中要处理的字段名，如 social_hotspots（必填）")
    parser.add_argument("img_dir", nargs="?", default="", help="图片保存目录（--summary-only 时可省略）")
    parser.add_argument("--img-only", action="store_true", help="仅下载图片，不处理summary/focus_point等")
    parser.add_argument("--summary-only", action="store_true", help="仅生成summary和focus_point，不下载图片")
    args = parser.parse_args()
    if args.img_only and args.summary_only:
        print("❌ --img-only 和 --summary-only 不能同时使用")
        sys.exit(1)
    if not args.img_dir and not args.summary_only:
        print("❌ 非 --summary-only 模式时，img_dir 为必填参数")
        sys.exit(1)

    # 支持多个key用逗号分隔
    keys = [k.strip() for k in args.key.split(',') if k.strip()]
    if len(keys) > 1:
        import subprocess, sys
        base = [sys.executable, '-u', __file__, args.input]
        for k in keys:
            cmd = base + [k]
            if args.img_dir: cmd.append(args.img_dir)
            if args.img_only: cmd.append('--img-only')
            if args.summary_only: cmd.append('--summary-only')
            r = subprocess.run(cmd)
            if r.returncode != 0:
                sys.exit(r.returncode)
        sys.exit(0)

    in_path = args.input
    out_path = args.input
    json_key = args.key
    img_dir = args.img_dir

    print(f"[{datetime.now().strftime('%H:%M:%S')}] 提取真实文章链接和封面图")
    print(f"  输入: {in_path}")
    _load_learn_focus_titles(in_path)

    with open(in_path, 'r', encoding='utf-8') as f:
        data = json.load(f)

    if isinstance(data, dict):
        spots = data.get(json_key, [])
    elif isinstance(data, list):
        spots = []
        for item in data:
            if item.get('name') == json_key:
                spots = item.get('list', [])
                break
    else:
        print("❌ 未知数据结构")
        sys.exit(1)

    print(f"  {json_key}: {len(spots)} 条")

    # 找出需要处理的条目
    need_proc = []
    need_summary = []
    for i, item in enumerate(spots):
        url = item.get('source_url', '')
        thumb = item.get('thumb', '')
        if not url:
            continue
        # 需要处理的条件：无thumb / 外链thumb（防盗链） / http/https链接需下载到本地
        if not thumb or thumb.startswith('http://') or thumb.startswith('https://') or 'article.zlink' in thumb or '1x1.gif' in thumb:
            need_proc.append((i, item))
        # 所有条目：summary/focus_point 为空或不符合格式即自动生成
        s = item.get('summary', '')
        f = item.get('focus_point', '')
        if not s or not f or len(s) > 40 or not s.endswith('。') or len(f) > 35:
            need_summary.append((i, item))

    if args.summary_only:
        if not need_summary:
            print("  ✅ 无需处理summary")
    if args.summary_only:
        if not need_summary:
            print("  ✅ 无需处理summary")
        else:
            print(f"\n=== 仅生成summary和focus_point ({len(need_summary)}条) ===")
            try:
                from playwright.sync_api import sync_playwright as _pw
                with _pw() as _pw_ctx:
                    _browser = _pw_ctx.chromium.launch(executable_path=CHROME_PATH, headless=True, args=["--no-sandbox"])
                    for idx, (si, sitem) in enumerate(need_summary):
                        url = sitem.get('source_url', '')
                        if 'so.toutiao.com/search' in url or 's.weibo.com/weibo' in url:
                            try:
                                _p = _browser.new_page()
                                _p.goto(url, wait_until="domcontentloaded", timeout=20000)
                                _p.wait_for_timeout(3000)
                                _real = _p.evaluate("""() => {
                                    const links = document.querySelectorAll("a");
                                    for (const a of links) {
                                        const h = a.href || "";
                                        if (h.includes("m.toutiao.com/group/") || h.includes("toutiaoimg.cn/group/")) return h;
                                    }
                                    for (const a of links) {
                                        const h = a.href || "";
                                        if (h.includes("toutiao.com/group/") && !h.includes("search/jump")) return h;
                                    }
                                    return "";
                                }""")
                                if _real:
                                    _real = _real.split("?")[0] if "?" in _real else _real
                                    sitem["source_url"] = _real
                                    print(f"      \u26a0 \u66f4\u65b0 source_url: {_real[:60]}...")
                                _p.close()
                            except:
                                pass
                        gen_summary_and_focus(sitem, sitem.get("source_url", ""), True)
                    _browser.close()
            except:
                pass
        with open(out_path, 'w', encoding='utf-8') as f:
            json.dump(data, f, ensure_ascii=False, indent=2)
        print(f"\n✅ 已保存到 {out_path}")
        return
        return

    print(f"  需处理: {len(need_proc)} 条")

    mmdd = os.path.basename(img_dir.rstrip('/'))
    os.makedirs(img_dir, exist_ok=True)
    print(f"  图片目录: {img_dir}")

    with sync_playwright() as pw:
        browser = pw.chromium.launch(executable_path=CHROME_PATH, headless=True, args=["--no-sandbox"])
        for idx, (i, item) in enumerate(need_proc):
            proc_item(item, browser, img_dir, mmdd, i, json_key, args.img_only)
        browser.close()

    # 图片验证：横向构图检查 + 缩放到≤750px宽（仅 --img-only 时）
    if args.img_only and need_proc:
        print(f"\n🔍 检查图片构图...")
        checked = 0
        fixed = 0
        for idx, (i, item) in enumerate(need_proc):
            fname = f"{json_key}_{i}.jpg"
            local_path = os.path.join(img_dir, fname)
            if not os.path.exists(local_path):
                continue
            try:
                from PIL import Image
                with Image.open(local_path) as img:
                    w, h = img.size
                    aspect = w / h if h > 0 else 0
                    orientation = '横向' if aspect >= 1.2 else '竖向'
                    if aspect > 3.0:
                        orientation = '超宽banner'
                    print(f"  [{i}] {w}x{h} ({aspect:.2f}) {orientation}: {item.get('title','')[:30]}", end='')
                    checked += 1
                    # 非横向(竖向或超宽banner) → 用豆包重新生成
                    need_regen = False
                    if aspect < 1.2 or aspect > 3.0:
                        need_regen = True
                    if need_regen:
                        title = item.get('title', '')
                        summary = item.get('summary', '')
                        # 豆包配图规则（2026-09-02 用户新增）：提示词不体现人物，只关注事件与品牌
                        prompt = f"{title}，{summary}，新闻配图风格，写实，横向构图，16:9，画面不出现任何人物姓名与肖像，只呈现事件场景和品牌产品元素"
                        import urllib.request as _url_req2
                        _payload = json.dumps({"model":"doubao-seedream-4-0-250828","prompt":prompt,"size":"1280x720","n":1}).encode()
                        _req2 = _url_req2.Request("https://ark.cn-beijing.volces.com/api/v3/images/generations", data=_payload,
                            headers={"Content-Type":"application/json","Authorization":f"Bearer {DOUBAO_API_KEY}"})
                        _resp2 = _url_req2.urlopen(_req2, timeout=60)
                        _img_url = json.loads(_resp2.read().decode("utf-8"))["data"][0]["url"]
                        _url_req2.urlretrieve(_img_url, local_path)
                        print(f" → 重新生成 ✅ 1280x720")
                        fixed += 1
                    else:
                        print()
                    # 缩放到≤750px宽
                    if w > 750:
                        ratio = 750 / w
                        new_h = int(h * ratio)
                        img_resized = img.resize((750, new_h), Image.LANCZOS)
                        img_resized.save(local_path, "JPEG", quality=85)
            except:
                pass
        print(f"\n检查: {checked} 张, 修复: {fixed} 张")

    # 处理summary/focus_point（--img-only 时不处理）
    if need_summary and not args.img_only:
        print(f"\n生成summary和focus_point ({len(need_summary)}条)...")
        for idx, (i, item) in enumerate(need_summary):
            url = item.get('source_url', '')
            gen_summary_and_focus(item, url, True)

    with open(out_path, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print(f"\n✅ 已保存到 {out_path}")


if __name__ == '__main__':
    main()
