#!/usr/bin/python3.12
"""从Dify工作流获取指定catName的热点内容，写入 origin_data/{MMDD}data.json 对应模块

用法:
    python3 get_brand_from_dify_v3.py <catName>

参数:
    <catName>  Dify 的 catName 值；同时是 origin_data JSON 中模块(section)的 name。
               例: brand_hotspots / vehicle_hotspots

说明:
    - 兼容两种响应结构: 直接条目数组 / section 包装([{name,opinion,list}])
    - 写入时按模块 name 清空旧数据后替换，模块不存在则新建
"""

import json, os, sys
from datetime import datetime
import urllib.request

# ── 配置 ──────────────────────────────────────────────────────
DIFY_URL = "http://43.164.190.156/v1/workflows/run"
DIFY_TOKEN = "app-ODwFQHiS45BuPerRlw77PVv8"
TODAY = datetime.now()
MMDD = TODAY.strftime('%m%d')
ORIGIN_PATH = f'/data/news/json/origin_data/{MMDD}data.json'

# ── 工具 ──────────────────────────────────────────────────────
def load_origin():
    if os.path.exists(ORIGIN_PATH):
        with open(ORIGIN_PATH, 'r', encoding='utf-8') as f:
            return json.load(f)
    return []

def save_origin(data):
    os.makedirs(os.path.dirname(ORIGIN_PATH), exist_ok=True)
    with open(ORIGIN_PATH, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)

def call_dify(cat_name):
    """调用Dify工作流（blocking模式），返回解析后的条目列表"""
    payload = json.dumps({
        "inputs": {"method": "GET", "catName": cat_name, "input": ""},
        "response_mode": "blocking",
        "user": "root"
    }).encode('utf-8')

    req = urllib.request.Request(
        DIFY_URL, data=payload,
        headers={
            'Authorization': f'Bearer {DIFY_TOKEN}',
            'Content-Type': 'application/json'
        },
        method='POST'
    )

    try:
        resp = urllib.request.urlopen(req, timeout=600)
    except Exception as e:
        raise ConnectionError(f"Dify请求失败: {e}")

    raw = resp.read().decode('utf-8', errors='replace')

    # 解析顶层 JSON
    top = json.loads(raw)
    # data.outputs.output 是 JSON 字符串（可能是条目数组，也可能是 section 包装 [{name,opinion,list}])
    output_str = top.get('data', {}).get('outputs', {}).get('output', '[]')
    # 解析为 Python 列表
    parsed = json.loads(output_str)

    if not isinstance(parsed, list):
        raise ValueError(f"解析结果非列表: {type(parsed)}")

    # 展开 section 包装结构：若元素含 'list' 字段则取 list 合并，否则视为条目数组
    items = []
    for item in parsed:
        if isinstance(item, dict) and isinstance(item.get('list'), list):
            items.extend(item['list'])
        else:
            items.append(item)

    print(f"  ✅ Dify 返回 {len(items)} 条 {cat_name} 内容")
    return items

def filter_items(items):
    """保留有真实标题的条目（新接口 brand 字段大量为空属常态，不按 brand 过滤）"""
    filtered = []
    for i in items:
        if not isinstance(i, dict):
            continue
        title = i.get('title', '')
        if title:
            if 'yesorno' not in i:
                i['yesorno'] = ''
            filtered.append(i)
    return filtered

def replace_section(origin_data, cat_name, new_list):
    """找到 name==cat_name 的段，清空后替换为新数据；
       若不存在则新建"""
    for section in origin_data:
        if section.get('name') == cat_name:
            old_count = len(section.get('list', []))
            section['list'] = new_list
            print(f"  🔄 已清空旧数据({old_count}条)，写入{len(new_list)}条")
            return origin_data
    # 不存在则新建
    origin_data.append({
        "name": cat_name,
        "opinion": "",
        "list": new_list
    })
    print(f"  🆕 新建 {cat_name} 段，写入{len(new_list)}条")
    return origin_data

def main():
    if len(sys.argv) < 2 or not sys.argv[1].strip():
        print("❌ 缺少参数 catName")
        print("   用法: python3 get_brand_from_dify_v3.py <catName>")
        print("   例:   python3 get_brand_from_dify_v3.py brand_hotspots")
        print("         python3 get_brand_from_dify_v3.py vehicle_hotspots")
        sys.exit(2)

    cat_name = sys.argv[1].strip()
    print(f'📅 日期: {TODAY.strftime("%Y-%m-%d")}, MMDD={MMDD}')
    print(f'📄 目标: {ORIGIN_PATH}  (模块: {cat_name})')
    print()

    # 1. 调用 Dify
    print(f'🔍 请求 Dify 工作流（catName={cat_name}）...')
    items = call_dify(cat_name)

    if not items:
        print('❌ Dify 返回为空，不写入')
        sys.exit(1)

    # 2. 过滤无标题条目
    items = filter_items(items)
    print(f'  🧹 过滤后保留 {len(items)} 条')

    # 2.5 现代动态 publish_time 只保留日期（2026-09-06 用户规则：不带具体时间点）
    if cat_name in ('hyundai_buzz_topics_domestic', 'hyundai_buzz_topics_international'):
        for it in items:
            pt = it.get('publish_time', '')
            if pt and len(pt) >= 10:
                it['publish_time'] = pt[:10]
        print(f'  🕐 现代动态 publish_time 已归一化为日期（{len(items)}条）')

    # 3. 加载 origin_data
    origin_data = load_origin()
    print(f'📂 已读取 origin_data，共 {len(origin_data)} 个段')

    # 4. 替换 cat_name 对应模块（清空 + 插入）
    origin_data = replace_section(origin_data, cat_name, items)

    # 5. 写入
    save_origin(origin_data)

    # 6. 统计
    for section in origin_data:
        if section.get('name') == cat_name:
            total = len(section['list'])
            print(f'\n📊 汇总: {cat_name} 共 {total} 条')
            break

    print(f'\n✅ 完成，已写入 {ORIGIN_PATH}')

if __name__ == '__main__':
    main()
