#!/usr/bin/env python3
"""
占位图自动修复脚本：检测 images/ 目录中的占位图，用 Doubao API 自动重新生成覆盖原路径。

已知占位图特征（两类）：
  1. 媒体默认占位图: 750x452 / 66592 字节（JPEG）
  2. 媒体来源占位图: 600x400 / 14570 字节（WEBP 伪装成 .jpg）

用法: python3 fix_placeholder_images.py <yes_data_json> <images_dir>
示例: python3 fix_placeholder_images.py /data/news/json/yes_data/0726data.json /data/news/images/0727/
"""
import json, os, sys, urllib.request

API_KEY = "ark-563a4210-d804-47fa-a37a-40962192119b-ce6db"
API_URL = "https://ark.cn-beijing.volces.com/api/v3/images/generations"
MODEL = "doubao-seedream-4-0-250828"

def get_title_map(json_path):
    """从yes_data获取文件名→标题映射"""
    titles = {}
    with open(json_path, 'r', encoding='utf-8') as f:
        data = json.load(f)
    items = data if isinstance(data, list) else [data]
    for section in items:
        if not isinstance(section, dict):
            continue
        for idx, item in enumerate(section.get('list', [])):
            fname = f"{section.get('name')}_{idx}.jpg"
            titles[fname] = item.get('title', '')
    return titles

def fix_image(img_path, title):
    """用Doubao API生成并覆盖占位图"""
    # 豆包配图规则（2026-09-02 用户新增）：提示词不体现人物，只关注事件与品牌
    # ⚠️ 2026-09-24 修复：人名须物理删除（见 person_name_filter.py）
    from person_name_filter import sanitize as _sanitize, NO_PERSON_SUFFIX as _NOPS
    title = _sanitize(title) or '该品牌与产品的营销活动现场'
    prompt = f"配图：{title}，新闻配图风格，写实，横向构图，16:9，高清" + _NOPS
    payload = json.dumps({
        "model": MODEL, "prompt": prompt,
        "size": "1280x720", "n": 1, "response_format": "url"
    }).encode()
    req = urllib.request.Request(API_URL, data=payload,
        headers={'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'})
    
    try:
        resp = urllib.request.urlopen(req, timeout=120)
        result = json.loads(resp.read().decode('utf-8'))
        img_url = result.get('data', [{}])[0].get('url', '')
        if not img_url:
            return False
        img_data = urllib.request.urlopen(
            urllib.request.Request(img_url, headers={'User-Agent': 'Mozilla/5.0'}),
            timeout=60
        ).read()
        with open(img_path, 'wb') as f:
            f.write(img_data)
        return True
    except Exception as e:
        print(f"  ❌ 生成失败: {e}")
        return False

def main():
    if len(sys.argv) != 3:
        print("用法: python3 fix_placeholder_images.py <yes_data_json> <images_dir>")
        sys.exit(1)
    
    json_path = sys.argv[1]
    img_dir = sys.argv[2]
    
    if not os.path.exists(img_dir):
        print(f"目录不存在: {img_dir}")
        sys.exit(1)
    
    titles = get_title_map(json_path)
    fixed = 0
    
    print(f"🔍 扫描占位图: {img_dir}")
    for fname in sorted(os.listdir(img_dir)):
        if not fname.endswith('.jpg') or fname.endswith('.png'):
            continue
        
        path = os.path.join(img_dir, fname)
        stat = os.stat(path)
        
        # 已知占位图特征（严格匹配，防止误杀正常图片）：
        #   1. 媒体默认占位图: 750x452 / 66592字节
        #   2. 媒体来源占位图: 600x400 / 14570字节（WEBP伪装jpg）
        is_placeholder_size = stat.st_size in (66592, 14570)
        if not is_placeholder_size:
            continue
        
        try:
            from PIL import Image
            im = Image.open(path)
            is_placeholder = (
                (stat.st_size == 66592 and im.size == (750, 452)) or
                (stat.st_size == 14570 and im.size == (600, 400))
            )
            if is_placeholder:
                title = titles.get(fname, fname.replace('.jpg', ''))
                print(f"  ⚠️ 占位图: {fname} ({title[:30]}) → 重新生成...")
                if fix_image(path, title):
                    new_sz = os.path.getsize(path)
                    print(f"    ✅ 修复完成: {new_sz}字节")
                    fixed += 1
        except Exception:
            continue
    
    print(f"\n✅ 共修复 {fixed} 张占位图")
    return 0

if __name__ == '__main__':
    sys.exit(main())
