#!/usr/bin/env python3.12
"""恢复 merge_yes_learn.py 清空的 yesorno（按标题从 learn_records 回填）

背景：merge_yes_learn.py **每次运行必然清空 yesorno**（已知 bug，非偶发，见
      SKILL.md 铁律与 references/merge_yes_learn_bug.md）。
本脚本以 learn_records 底稿为准，按 title 精确匹配回填 yesorno，
并**保持原文件的 list/dict 外层结构不变**。

用法：
  python3.12 restore_yesorno.py <learn_file> <yes_data_file>
"""
import json
import sys


def load_raw(p):
    return json.load(open(p, encoding='utf-8'))


def sections_of(raw):
    """统一取 {section_name: [entries]}"""
    if isinstance(raw, list):
        return {s['name']: s['list'] for s in raw
                if isinstance(s, dict) and 'name' in s}
    return {k: v for k, v in raw.items() if isinstance(v, list)}


def main():
    learn_p, yes_p = sys.argv[1], sys.argv[2]
    learn_raw = load_raw(learn_p)
    raw = load_raw(yes_p)

    learn = sections_of(learn_raw)
    yes = sections_of(raw)

    want = set()
    for lst in learn.values():
        for e in lst:
            if str(e.get('yesorno', '')).strip().lower() == 'yes':
                want.add(str(e.get('title', '')).strip())

    restored, missing, total = 0, [], 0
    for sec, lst in yes.items():
        for e in lst:
            total += 1
            t = str(e.get('title', '')).strip()
            if t in want:
                e['yesorno'] = 'yes'
                restored += 1
            else:
                missing.append((sec, t[:40]))

    json.dump(raw, open(yes_p, 'w', encoding='utf-8'),
              ensure_ascii=False, indent=2)

    print(f'底稿 yes 标题数: {len(want)}')
    print(f'yes_data 总条数: {total} → 回填 yesorno: {restored} 条')
    if missing:
        print(f'⚠️ 未匹配 {len(missing)} 条:')
        for s, t in missing:
            print(f'   [{s}] {t}')
    else:
        print('✅ 全部匹配')


if __name__ == '__main__':
    main()
