#!/usr/bin/env python3.12
"""vehicle_hotspots_3：从 autohome 正文取车辆照片（修复阶梯①）"""
import io, os, re, subprocess
from PIL import Image

IMG = '/data/news/images/0927/'
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126 Safari/537.36'
URL = 'https://www.autohome.com.cn/news/202609/1317447.html'


def fetch(url):
    r = subprocess.run(['curl', '-sSL', '--max-time', '20', '-A', UA,
                        '-e', 'https://www.autohome.com.cn/', url], capture_output=True)
    return r.stdout if r.returncode == 0 else None


page = fetch(URL)
print('页面字节:', len(page) if page else 0)
html = page.decode('utf-8', 'ignore') if page else ''

# autohome 正文图：chejiahaodfs / autohomecar
urls = re.findall(r'(https?://[a-z0-9.]*autoimg\.cn/[^\s"\']+?\.(?:jpg|jpeg|png))', html)
urls = list(dict.fromkeys(urls))
print('发现图片候选:', len(urls))
for u in urls[:12]:
    print('  ', u[:110])

# 优先取大图变体
picked = None
for u in urls:
    if 'logo' in u.lower() or 'icon' in u.lower():
        continue
    big = re.sub(r'\d+x\d+_\d+', '1600x0_1', u)
    if big == u and 'x0_1' not in u:
        continue
    picked = big
    break
print('\n选定:', picked)
if picked:
    data = fetch(picked)
    print('下载字节:', len(data) if data else 0)
    if data and len(data) > 2000:
        im = Image.open(io.BytesIO(data))
        src = f'{im.format} {im.size[0]}x{im.size[1]}'
        if im.mode in ('RGBA', 'LA', 'P'):
            im = im.convert('RGBA')
            bg = Image.new('RGB', im.size, (255, 255, 255))
            bg.paste(im, mask=im.split()[-1])
            im = bg
        else:
            im = im.convert('RGB')
        if im.width >= 600:
            if im.width > 750:
                im = im.resize((750, round(im.height * 750 / im.width)), Image.LANCZOS)
            out = IMG + 'vehicle_hotspots_3.jpg'
            im.save(out, 'JPEG', quality=92)
            print(f'  ✅ {src} → JPEG {im.size} {os.path.getsize(out)//1024}KB')
        else:
            print(f'  ❌ 宽度不足: {src}')
