#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""register_jcx.py —— 注册 crawl_jcx_tzgg.py"""
import json
import os
import shutil
import sys

CFG = "/root/gov_crawler/daily_crawl_config.json"
BAK = "/root/gov_crawler/Archive/daily_crawl_config.json.bak_20260924_jcx"
SCRIPT = "crawl_jcx_tzgg.py"

cfg = json.load(open(CFG, encoding="utf-8"))
print("注册前条目数:", len(cfg))
if any((c.get("script") or "") == SCRIPT for c in cfg):
    print("⚠️ 已注册")
    sys.exit(0)

tmpl = None
for c in cfg:
    if (c.get("group") or "") == "云南":
        tmpl = c
        break
if tmpl is None:
    tmpl = cfg[-1]
print("模板键:", sorted(tmpl.keys()))
print("模板样例:", json.dumps(tmpl, ensure_ascii=False)[:240])

entry = dict(tmpl)
entry.update({
    "name": "江城县人民政府-通知公告",
    "display_name": "江城县人民政府-通知公告",
    "script": SCRIPT,
    "args": ["--pages=5"],
    "group": "云南",
    "note": "云南普洱江城哈尼族彝族自治县 通知公告, TRS vsb 静态列表(line_u9_N), "
            "33页492条/15条每页, 分页倒序静态链接须跟随「下页」(第N页=tzgg/{34-N}.htm, "
            "页脚另有「尾页」也是class=Next须按文字取), 列表 a[title]+b.date(文字可能截断优先title属性), "
            "正文 div.article_detail, 附件为正文内 virtual_attach_file.vsb?afc= 无扩展名(须专判)",
})
print("\n待追加:", json.dumps(entry, ensure_ascii=False)[:420])

cfg.append(entry)
os.makedirs(os.path.dirname(BAK), exist_ok=True)
if not os.path.exists(BAK):
    shutil.copy2(CFG, BAK)
with open(CFG, "w", encoding="utf-8") as f:
    json.dump(cfg, f, ensure_ascii=False, indent=2)

cfg2 = json.load(open(CFG, encoding="utf-8"))
print("\n注册后条目数:", len(cfg2))
missing = [c.get("script") for c in cfg2
           if c.get("script") and not os.path.exists(os.path.join("/root/gov_crawler", c["script"]))]
print("config 中 script 磁盘缺失:", len(missing), missing[:5])
idx = [i + 1 for i, c in enumerate(cfg2) if (c.get("script") or "") == SCRIPT]
print("✅ 已注册 %s 序号 = %s" % (SCRIPT, idx))
print("备份:", BAK)
