#!/usr/bin/python3.12
"""从Dify现代热点工作流获取数据，写入origin_data hyundai_buzz_topics_domestic"""
import json, os, sys, urllib.request, http.client
from datetime import datetime, timedelta
# 每日营销日报-现代社媒动态
WORKFLOW_URL = "http://43.164.190.156/v1/workflows/run"
WORKFLOW_TOKEN = "app-f9e3eEzqZkR1tJdFczEhRFUL"

def call_workflow():
    payload = json.dumps({
        "inputs": {},
        "response_mode": "blocking",
        "user": "root"
    }).encode()

    req = urllib.request.Request(WORKFLOW_URL, data=payload,
        headers={'Authorization': f'Bearer {WORKFLOW_TOKEN}', 'Content-Type': 'application/json'},
        method='POST')

    try:
        with urllib.request.urlopen(req, timeout=600) as r:
            return json.loads(r.read().decode('utf-8'))
    except urllib.error.HTTPError as e:
        raise Exception(f"工作流失败 HTTP {e.code}: {e.read().decode('utf-8')[:200]}")

def extract_hyundai(resp):
    """从响应中提取hyundai_hotspots（支持多种嵌套格式）"""
    data = resp.get('data', {})
    outputs = data.get('outputs', {})
    if not outputs:
        # 尝试直接取output字段
        output_field = data.get('output', data.get('result', {}))
        if isinstance(output_field, dict):
            outputs = output_field
        else:
            return None
    
    # 各层级取值
    val = outputs.get('hyundai_hotspots')
    if val is None:
        for k in ['output', 'result', 'data']:
            sub = outputs.get(k)
            if isinstance(sub, dict):
                val = sub.get('hyundai_hotspots')
                if val:
                    break
    
    # 尝试解析各种格式
    if isinstance(val, str):
        try:
            val = json.loads(val)
        except:
            print(f"  ⚠️ hyundai_hotspots 无法解析为JSON: {val[:100]}")
            return None
    
    if isinstance(val, list):
        result = [i for i in val if isinstance(i, dict)]
        print(f"  提取到 {len(result)} 条现代热点")
        return result
    
    # 如果是dict套list
    if isinstance(val, dict):
        for v in val.values():
            if isinstance(v, list):
                result = [i for i in v if isinstance(i, dict)]
                if result:
                    print(f"  提取到 {len(result)} 条现代热点（嵌套dict）")
                    return result
    
    print(f"  ⚠️ 无法解析的hyundai_hotspots类型: {type(val).__name__}")
    return None

def merge_to_origin(mmd_data, items, seen_titles):
    """追加到hyundai_buzz_topics_domestic，按title去重"""
    added = 0
    for item in items:
        t = item.get('title', '')
        if t and t not in seen_titles:
            seen_titles.add(t)
            # 确保包含必要字段
            entry = {
                'brand': item.get('brand', ''),
                'model': item.get('model', ''),
                'title': item.get('title', ''),
                'summary': item.get('summary', ''),
                'focus_point': item.get('focus_point', ''),
                'publish_time': item.get('publish_time', ''),
                'thumb': item.get('thumb', ''),
                'source_url': item.get('source_url', ''),
                'platform': item.get('platform', ''),
                'origin_url': item.get('origin_url', ''),
                'icon': item.get('icon', ''),
                'heat_score': item.get('heat_score', 0),
            }
            added += 1
            # 找到hyundai_buzz_topics_domestic并插入
            found = False
            for entry_item in mmd_data:
                if entry_item.get('name') == 'hyundai_buzz_topics_domestic':
                    entry_item['list'].append(entry)
                    found = True
                    break
            if not found:
                mmd_data.append({"name": "hyundai_buzz_topics_domestic", "opinion": "", "list": [entry]})
    return added

def main():
    mmdd = (datetime.now() + timedelta(days=1)).strftime("%m%d")
    origin_path = f"/data/news/json/origin_data/{mmdd}data.json"
    if not os.path.exists(origin_path):
        mmdd = datetime.now().strftime("%m%d")
        origin_path = f"/data/news/json/origin_data/{mmdd}data.json"
    print(f"[{datetime.now().strftime('%H:%M:%S')}] 获取现代热点")
    print(f"  origin: {origin_path}")

    # 调用工作流
    resp = call_workflow()
    status = resp.get('data', {}).get('status')
    elapsed = resp.get('data', {}).get('elapsed_time', 0)
    print(f"  status: {status}, elapsed: {elapsed:.1f}s")

    if status != 'succeeded':
        print(f"❌ 工作流失败: {resp.get('data', {}).get('error', 'unknown')}")
        sys.exit(1)

    # 提取数据
    items = extract_hyundai(resp)
    if not items:
        print("❌ 未提取到 hyundai_hotspots")
        print(f"  outputs keys: {list(resp.get('data',{}).get('outputs',{}).keys())}")
        sys.exit(1)
    print(f"  提取到 {len(items)} 条")

    # 日期过滤：只保留今天和昨天的数据（兼容纯日期和含时间的格式）
    today_str = datetime.now().strftime('%Y-%m-%d')
    yesterday_str = (datetime.now() - timedelta(days=1)).strftime('%Y-%m-%d')
    allowed_dates = {today_str, yesterday_str}
    before = len(items)
    items = [i for i in items if i.get('publish_time', '')[:10] in allowed_dates]
    after = len(items)
    if before - after > 0:
        print(f"  日期过滤: 移除 {before - after} 条非24h数据")

    # 读取origin_data
    mmd_data = []
    seen_titles = set()
    if os.path.exists(origin_path):
        with open(origin_path, 'r', encoding='utf-8') as f:
            mmd_data = json.load(f)
        # 已有数据的title去重
        for entry in mmd_data:
            if entry.get('name') == 'hyundai_buzz_topics_domestic':
                for it in entry.get('list', []):
                    t = it.get('title', '')
                    if t:
                        seen_titles.add(t)
    else:
        print("  ⚠️ 新建文件")

    # 合并写入
    if items:
        added = merge_to_origin(mmd_data, items, seen_titles)
        print(f"  新增 {added} 条（跳过 {len(items)-added} 条重复）")
    else:
        print("  Dify 无数据，跳过合并")

    os.makedirs(os.path.dirname(origin_path), exist_ok=True)
    with open(origin_path, 'w', encoding='utf-8') as f:
        json.dump(mmd_data, f, ensure_ascii=False, indent=2)

    total = 0
    for entry in mmd_data:
        if entry.get('name') == 'hyundai_buzz_topics_domestic':
            total = len(entry.get('list', []))
    print(f"✅ 已写入 {origin_path}（hyundai_buzz_topics_domestic共{total}条）")

    # 从vehicle_hotspots迁移现代/起亚相关条目
    hyundai_kw = ['现代', '起亚', '捷尼赛思', '摩比斯']
    for entry in mmd_data:
        if entry.get('name') not in ('vehicle_hotspots', 'vehicle_hotposts'):
            continue
        vehicle_list = entry.get('list', [])
        # 找到hyundai_domestic的现有标题
        hyundai_titles = set()
        for he in mmd_data:
            if he.get('name') == 'hyundai_buzz_topics_domestic':
                hyundai_titles = {e.get('title', '') for e in he.get('list', [])}
                break
        moved = 0
        keep = []
        for item in vehicle_list:
            t = item.get('title', '')
            if any(kw in t for kw in hyundai_kw):
                if t not in hyundai_titles:
                    # 迁移到hyundai_domestic
                    for he in mmd_data:
                        if he.get('name') == 'hyundai_buzz_topics_domestic':
                            item['yesorno'] = ''
                            he['list'].append(dict(item))
                            hyundai_titles.add(t)
                            moved += 1
                            print(f'  🚚 从vehicle迁移: {t[:40]}')
                            break
            else:
                keep.append(item)
        vehicle_list[:] = keep
        if moved:
            print(f'  共迁移 {moved} 条到hyundai_buzz_topics_domestic')
        break

    # 再保存一次（迁移后）
    with open(origin_path, 'w', encoding='utf-8') as f:
        json.dump(mmd_data, f, ensure_ascii=False, indent=2)

if __name__ == '__main__':
    main()
