#!/usr/bin/env python3.12
# -*- coding: utf-8 -*-
"""第二阶段：把第一阶段拉取的原始数据拷贝到 learn_records，供用户手动标注。

用法:
    python3.12 /data/news/scripts/copy_origin_to_learn.py [MMDD]

源:   /data/news/json/origin_data/{MMDD}data.json   （第一阶段 5 条命令的产出）
目标: /data/news/learn_records/{MMDD}data_1.json    （用户手动标注 yesorno / focus_point）

标注方式：把要进日报的条目标成  "yesorno": "yes" ，并补齐 focus_point（策略启示）。

⚠️ 默认**保留**源文件自带的 yesorno / focus_point（Dify 工作流会按 5/5/5/3/3 配额直接返回
   yesorno="yes"+focus_point，0915/0916 底稿均带 21 条 yes）。旧「清空」行为需显式 --clear-yes。
   （2026-09-17 用户纠正：此前的默认清空属回归 bug，会让标注凭空消失。）
标注完成后即可进入第三阶段：merge_yes_learn 会以 learn_records 为准提取 yes 条目。

⚠️ 铁律：learn_records/ 是用户手动标注目录，本脚本**绝不覆盖**任何已有文件。
   目标已存在 → 直接报错退出，不动任何文件（列出已存在的文件供人工处理）。
   确需另建一份时用 --force，会写成 _2/_3 递增序号，仍然不覆盖既有文件。
"""
import argparse
import glob
import json
import os
import shutil
import sys
from datetime import datetime

ORIGIN_DIR = "/data/news/json/origin_data"
LEARN_DIR = "/data/news/learn_records"
SECTIONS = ("brand_hotspots", "vehicle_hotspots", "social_hotspots",
            "hyundai_buzz_topics_domestic", "hyundai_buzz_topics_international")


def main():
    ap = argparse.ArgumentParser(description="拷贝 origin_data → learn_records（供手动标注）")
    ap.add_argument("mmdd", nargs="?", default=None, help="日期 MMDD（默认取当天）")
    ap.add_argument("--force", action="store_true",
                    help="已有 {MMDD}data*.json 时仍另建一份（写 _N 递增序号，绝不覆盖）")
    ap.add_argument("--clear-yes", action="store_true",
                    help="旧行为：清空源数据自带的 yesorno（默认**保留**，见 2026-09-17 用户纠正）")
    args = ap.parse_args()

    mmdd = args.mmdd or datetime.now().strftime("%m%d")
    src = os.path.join(ORIGIN_DIR, "%sdata.json" % mmdd)
    if not os.path.exists(src):
        print("❌ 源文件不存在: %s" % src)
        print("   请先执行第一阶段 5 条命令（get_brand_from_dify_v3.py ...）")
        sys.exit(1)

    existing = sorted(glob.glob(os.path.join(LEARN_DIR, "%sdata*.json" % mmdd)))
    if existing and not args.force:
        print("❌ learn_records 中已存在 %s 的标注文件，**拒绝覆盖**（铁律：该目录为用户手动标注，禁止自动覆盖）:" % mmdd)
        for f in existing:
            print("   - %s" % os.path.basename(f))
        print("\n如确认要另建一份标注底稿，请加 --force（会写 _N 递增序号，不覆盖上面任何文件）")
        sys.exit(2)

    # 目标文件名：无既有文件 → _1；--force → 递增到下一个未占用序号
    idx = 1
    if existing:
        used = set()
        for f in existing:
            base = os.path.basename(f)
            try:
                used.add(int(base.replace("%sdata_" % mmdd, "").replace(".json", "").split("_")[0]))
            except ValueError:
                continue
        while idx in used:
            idx += 1
    target = os.path.join(LEARN_DIR, "%sdata_%d.json" % (mmdd, idx))
    while os.path.exists(target):
        idx += 1
        target = os.path.join(LEARN_DIR, "%sdata_%d.json" % (mmdd, idx))

    with open(src, encoding="utf-8") as f:
        data = json.load(f)

    # 2026-09-17 用户纠正（回归修复）：
    # 默认**保留**源数据自带的 yesorno / focus_point。Dify 工作流会按 5/5/5/3/3 配额
    # 直接返回 yesorno="yes" + focus_point，历史底稿（0915/0916data_1）均带 21 条 yes
    # 进入 learn_records。此前的「清空 yesorno」逻辑属回归 bug，会让用户看到的标注凭空消失。
    # 确需旧行为请显式加 --clear-yes。
    kept = 0
    if args.clear_yes:
        for sec in (data if isinstance(data, list) else []):
            for item in (sec.get("list") or []):
                if item.get("yesorno"):
                    item["yesorno"] = ""
    else:
        for sec in (data if isinstance(data, list) else []):
            for item in (sec.get("list") or []):
                if str(item.get("yesorno", "")).lower() == "yes":
                    kept += 1

    # 先写临时文件再原子落盘，避免半截文件
    tmp = target + ".tmp"
    with open(tmp, "w", encoding="utf-8") as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    os.replace(tmp, target)

    print("📅 MMDD = %s" % mmdd)
    print("📥 源:   %s (%d 字节)" % (src, os.path.getsize(src)))
    print("📤 目标: %s (%d 字节)" % (target, os.path.getsize(target)))
    if args.clear_yes:
        print("🧹 --clear-yes 已指定：yesorno 全部清空（旧行为）")
    else:
        print("✅ 已保留源数据自带的 yesorno / focus_point：yes = %d 条" % kept)
    print("\n📊 各板块条数：")
    total = 0
    for sec in (data if isinstance(data, list) else []):
        n = len(sec.get("list") or [])
        total += n
        print("   %-34s %3d 条" % (sec.get("name"), n))
    print("   %-34s %3d 条" % ("合计", total))

    print("\n" + "=" * 58)
    print("下一步（人工标注）：")
    print("  1. 打开 %s" % target)
    print('  2. 把要进日报的条目标成  "yesorno": "yes"')
    print("  3. 为这些条目写好 focus_point（策略启示，≤35字，用「可借鉴/可关注/不妨/建议」）")
    print("  4. 标注完成后执行第三阶段（merge_yes_learn 以 learn_records 为准）：")
    print("     LEARN_FILE=$(ls -t %s/%sdata_*.json | head -1)" % (LEARN_DIR, mmdd))
    print('     python3.12 /data/news/scripts/merge_yes_learn.py "$LEARN_FILE" /data/news/json/yes_data/%sdata.json' % mmdd)
    print("=" * 58)


if __name__ == "__main__":
    main()
