#!/usr/bin/env python3.12
"""配图修复阶梯：备份转 JPEG + 源站原图变体 + 正文取图"""
import io, os, re, subprocess
from PIL import Image

IMG = '/data/news/images/0927/'
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126 Safari/537.36'
REFS = {'bitautoimg': 'https://news.yiche.com/', 'autoimg': 'https://www.autohome.com.cn/',
        'ithome': 'https://www.ithome.com/', 'socialbeta': 'https://socialbeta.com/'}

TARGETS = {
    'vehicle_hotspots_1.jpg': [
        'https://image.bitautoimg.com/appimage-1024-w0/news/2026/09/26/44e40780-2282-4de9-91e4-17ec5174ef97.jpg',
        'https://image.bitautoimg.com/appimage-800-w0/news/2026/09/26/44e40780-2282-4de9-91e4-17ec5174ef97.jpg',
    ],
    'vehicle_hotspots_2.jpg': [
        'http://www2.autoimg.cn/chejiahaodfs/g33/M09/CF/D3/1600x0_1_autohomecar__Chto52q3Nc2AM533AC04CtkZ34M927.png',
        'http://img3.autoimg.cn/chejiahaodfs/g33/M09/CF/D3/1600x0_1_autohomecar__Chto52q3Nc2AM533AC04CtkZ34M927.png',
    ],
    'vehicle_hotspots_3.jpg': [
        'http://img3.autoimg.cn/chejiahaodfs/g33/M0B/D4/72/1600x0_1_autohomecar__ChxpVmq3uhCAZW-uABdHTzQNhsU283.png',
        'http://www2.autoimg.cn/chejiahaodfs/g33/M0B/D4/72/1600x0_1_autohomecar__ChxpVmq3uhCAZW-uABdHTzQNhsU283.png',
    ],
    'vehicle_hotspots_4.jpg': [
        'http://img3.autoimg.cn/chejiahaodfs/g34/M01/D0/DE/1600x0_1_autohomecar__ChxpV2q3KNmASlX6ABcwY6mcNqM374.png',
        'http://www2.autoimg.cn/chejiahaodfs/g34/M01/D0/DE/1600x0_1_autohomecar__ChxpV2q3KNmASlX6ABcwY6mcNqM374.png',
    ],
}


def fetch(url, ref=None):
    r = subprocess.run(['curl', '-sSL', '--max-time', '15', '-A', UA,
                        '-e', ref or 'https://www.baidu.com/', url], capture_output=True)
    return r.stdout if r.returncode == 0 else None


def save_jpeg(data, out, min_w=600, force=False):
    im = Image.open(io.BytesIO(data))
    src = f'{im.format} {im.size[0]}x{im.size[1]}'
    if im.mode in ('RGBA', 'LA', 'P'):
        im = im.convert('RGBA')
        bg = Image.new('RGB', im.size, (255, 255, 255))
        bg.paste(im, mask=im.split()[-1])
        im = bg
    else:
        im = im.convert('RGB')
    if im.width < min_w:
        return None, f'{src} → 宽度不足({im.width})'
    if im.width > 750:
        im = im.resize((750, round(im.height * 750 / im.width)), Image.LANCZOS)
    im.save(out, 'JPEG', quality=92)
    return f'{src} → JPEG {im.width}x{im.height} {os.path.getsize(out)//1024}KB', None


# ① 0 字节：优先 .png 备份转 JPEG
bak = IMG + 'brand_hotspots_1.jpg.png'
if os.path.exists(bak):
    msg, err = save_jpeg(open(bak, 'rb').read(), IMG + 'brand_hotspots_1.jpg')
    print(f"  {'✅' if msg else '❌'} brand_hotspots_1.jpg [备份转JPEG]: {msg or err}")

# ② 源站原图变体
for fn, urls in TARGETS.items():
    out = IMG + fn
    for u in urls:
        data = fetch(u, REFS.get(next((k for k in REFS if k in u), ''), None))
        if not data or len(data) < 2000:
            print(f'  ⚠️ {fn}: 取回 {len(data) if data else 0}B')
            continue
        try:
            msg, err = save_jpeg(data, out)
        except Exception as e:
            print(f'  ⚠️ {fn}: 解码失败({e})')
            continue
        if err:
            print(f'  ⚠️ {fn}: {err}')
            continue
        print(f'  ✅ {fn}: {msg}')
        break
    else:
        print(f'  ❌ {fn}: 全部失败 → 需豆包')

# ③ social_hotspots_2：ithome 正文取图
page = fetch('https://www.ithome.com/1/007/426.htm', REFS['ithome'])
if page:
    html = page.decode('utf-8', 'ignore')
    urls = re.findall(r'https://img\.ithome\.com/newsuploadfiles/[^\s"\']+?\.(?:jpg|jpeg|png)', html)
    urls = [u for u in dict.fromkeys(urls) if 'logo' not in u.lower()]
    print(f'  正文发现 {len(urls)} 张候选图')
    done = False
    for u in urls[:5]:
        data = fetch(u + '?x-bce-process=image/format,f_auto/resize,m_lfit,w_1200', REFS['ithome'])
        if not data or len(data) < 2000:
            continue
        try:
            msg, err = save_jpeg(data, IMG + 'social_hotspots_2.jpg')
        except Exception as e:
            continue
        if err:
            print(f'    ⚠️ {msg or err} ← {u[-40:]}')
            continue
        print(f'  ✅ social_hotspots_2.jpg: {msg}')
        done = True
        break
    if not done:
        print('  ❌ social_hotspots_2.jpg: 正文取图失败 → 需豆包')
else:
    print('  ❌ social_hotspots_2.jpg: 页面抓取失败')
