#!/usr/bin/python3.12
"""thumb图片本地化：遍历JSON中thumb字段，下载外链图片到本地"""
import json, os, urllib.request, re, sys
from datetime import datetime

MMDD = datetime.now().strftime('%m%d')
TARGET_DIR = f'/data/news/images/{MMDD}/'
os.makedirs(TARGET_DIR, exist_ok=True)

# 要处理的文件
files = [
    f'/data/news/json/{MMDD}data.json',
    f'/data/news/json/origin_data/{MMDD}data.json',
    f'/data/news/json/yes_data/{MMDD}data.json',
]

processed = 0
downloaded = 0

def process_items(items, section_name):
    global processed, downloaded
    if not isinstance(items, list):
        return
    for idx, item in enumerate(items):
        if not isinstance(item, dict):
            continue
        thumb = item.get('thumb', '') or ''
        if not thumb.startswith('http://') and not thumb.startswith('https://'):
            continue

        processed += 1
        # 确定文件扩展名
        ext = 'jpg'
        if '.png' in thumb.lower():
            ext = 'png'
        elif '.webp' in thumb.lower():
            ext = 'jpg'  # webp转jpg

        fname = f'{section_name}_{idx}.{ext}'
        local_path = os.path.join(TARGET_DIR, fname)

        try:
            req = urllib.request.Request(thumb, headers={'User-Agent': 'Mozilla/5.0'})
            resp = urllib.request.urlopen(req, timeout=15)
            data = resp.read()

            # webp转jpg
            if thumb.lower().endswith('.webp'):
                from PIL import Image
                import io
                img = Image.open(io.BytesIO(data)).convert('RGB')
                fname = f'{section_name}_{idx}.jpg'
                local_path = os.path.join(TARGET_DIR, fname)
                img.save(local_path, 'JPEG', quality=85)
            else:
                with open(local_path, 'wb') as f:
                    f.write(data)

            # 更新thumb路径
            item['thumb'] = f'/images/{MMDD}/{fname}'
            downloaded += 1
            print(f'  ✅ {thumb[:40]}... → {fname}')

        except Exception as e:
            print(f'  ⚠️ 下载失败: {thumb[:40]}... ({str(e)[:30]})')

for fpath in files:
    if not os.path.exists(fpath):
        continue
    try:
        data = json.load(open(fpath, encoding='utf-8'))
    except:
        continue

    if isinstance(data, list):
        for entry in data:
            if isinstance(entry, dict) and 'list' in entry:
                process_items(entry['list'], entry.get('name', 'unknown'))
    elif isinstance(data, dict):
        for sec in ['brand_hotspots', 'vehicle_hotspots', 'social_hotspots',
                     'hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international']:
            process_items(data.get(sec, []), sec)

    # 写回
    with open(fpath, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

print(f'\n处理完成: 扫描{processed}个外链, 下载{downloaded}张')
