#!/usr/bin/python3.12
"""
从 Dify 全量采集结果按 type 分类，输出到标准日报结构
用法: python3 categorize_from_dify.py <输入文件路径> <输出文件路径>
"""

import json, sys, os
from collections import OrderedDict
from datetime import datetime

SCHEME_PATH = "/data/news/json/origin_data/jsonScheme.json"

# type → 日报 section name 映射
TYPE_MAP = {
    "brand_marketing": "brand_hotspots",
    "auto_industry": "vehicle_hotspots",
    "social": "social_hotspots",
    "brand_news": "hyundai_buzz_topics_domestic",
}

# 输出 item 字段 → 输入 article 字段映射
FIELD_MAP = {
    "title": "title",
    "summary": "summary",
    "publish_time": "date",
    "platform": "media",
    "source_url": "url",
    "thumb": "image",
}

# 各 section 字段列表（从 jsonScheme 提取后覆盖）
SECTION_FIELDS = {
    "brand_hotspots": ["brand", "title", "summary", "focus_point", "source_url",
                       "publish_time", "platform", "thumb", "origin_url", "yesorno"],
    "vehicle_hotspots": ["brand", "title", "summary", "focus_point", "source_url",
                         "publish_time", "platform", "thumb", "origin_url", "yesorno"],
    "social_hotspots": ["brand", "title", "summary", "focus_point", "source_url",
                        "publish_time", "platform", "thumb", "origin_url", "yesorno",
                        "heat_score"],
    "hyundai_buzz_topics_domestic": ["brand", "title", "summary", "focus_point",
                                     "source_url", "publish_time", "platform",
                                     "thumb", "origin_url", "yesorno", "icon",
                                     "heat_score"],
    "hyundai_buzz_topics_international": ["brand", "title", "summary", "focus_point",
                                          "source_url", "publish_time", "platform",
                                          "thumb", "origin_url", "yesorno", "icon",
                                          "heat_score"],
}


def load_scheme_fields():
    """从 jsonScheme.json 提取各 section 的字段列表"""
    try:
        with open(SCHEME_PATH) as f:
            scheme = json.load(f)
        props = scheme.get("items", {}).get("properties", {})
        list_items = props.get("list", {}).get("items", {}).get("properties", {})
        return list(list_items.keys())
    except:
        return None


def make_item(article, fields):
    """创建一条 item，映射字段，不在映射里的留空"""
    item = OrderedDict()
    for f in fields:
        if f in FIELD_MAP:
            item[f] = article.get(FIELD_MAP[f], "")
        elif f == "heat_score":
            item[f] = 0  # 后续由 fill_heat_score 填充
        elif f == "icon":
            item[f] = ""
        elif f == "yesorno":
            item[f] = ""
        elif f == "brand":
            item[f] = ""
        elif f == "summary":
            item[f] = ""
        elif f == "focus_point":
            item[f] = ""
        elif f == "origin_url":
            item[f] = ""
        elif f == "thumb":
            item[f] = ""
        else:
            item[f] = article.get(f, "")
    return item


def main():
    if len(sys.argv) < 3:
        print("用法: python3 categorize_from_dify.py <输入文件> <输出文件>")
        sys.exit(1)

    in_path = sys.argv[1]
    out_path = sys.argv[2]

    if not os.path.exists(in_path):
        print(f"❌ 输入文件不存在: {in_path}")
        sys.exit(1)

    # 读取输入
    with open(in_path) as f:
        in_data = json.load(f)

    articles = in_data.get("articles", [])
    if not articles:
        print("⚠️ articles 为空")
        # 仍然输出空结构
        articles = []

    print(f"📄 共 {len(articles)} 条文章")

    # 读取 scheme 字段（可选，已经有默认值）
    scheme_fields = load_scheme_fields()
    if scheme_fields:
        # 用 scheme 的字段更新所有 section
        for sec in SECTION_FIELDS:
            # 保留 heat_score 和 icon 等特殊字段
            extras = set(SECTION_FIELDS.get(sec, [])) - set(scheme_fields)
            SECTION_FIELDS[sec] = list(scheme_fields) + list(extras)

    # 分类统计
    categorized = {}
    for article in articles:
        atype = article.get("type", "")
        sec = TYPE_MAP.get(atype)
        if not sec:
            continue
        if sec not in categorized:
            categorized[sec] = []
        fields = SECTION_FIELDS.get(sec, [])
        item = make_item(article, fields)
        categorized[sec].append(item)

    # 输出结构：按 jsonScheme 要求的数组格式 [{name, opinion, list}]
    from datetime import datetime, timedelta

    # scheme 中 vehicle_hotspots 名为 vehicle_hotposts
    scheme_sections = [
        ("brand_hotspots", "brand_hotspots"),
        ("vehicle_hotspots", "vehicle_hotspots"),
        ("social_hotspots", "social_hotspots"),
        ("hyundai_buzz_topics_domestic", "hyundai_buzz_topics_domestic"),
        ("hyundai_buzz_topics_international", "hyundai_buzz_topics_international"),
    ]

    output = []
    total = 0
    for sec_key, scheme_name in scheme_sections:
        items = categorized.get(sec_key, [])
        output.append({
            "name": scheme_name,
            "opinion": "",
            "list": items
        })
        total += len(items)
        print(f"  {scheme_name}: {len(items)} 条")

    # 写入
    os.makedirs(os.path.dirname(out_path) or ".", exist_ok=True)
    with open(out_path, "w", encoding="utf-8") as f:
        json.dump(output, f, ensure_ascii=False, indent=2)

    print(f"\n✅ 已保存 {out_path}（共 {total} 条）")


if __name__ == "__main__":
    main()
