#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""register_yingquan.py —— 注册 crawl_yingquan_gggs.py 到 daily_crawl_config.json"""
import json
import os
import shutil
import sys

CFG = "/root/gov_crawler/daily_crawl_config.json"
BAK = "/root/gov_crawler/Archive/daily_crawl_config.json.bak_20260924_yingquan"
SCRIPT = "crawl_yingquan_gggs.py"

cfg = json.load(open(CFG, encoding="utf-8"))
print("注册前条目数:", len(cfg))

if any((c.get("script") or "") == SCRIPT for c in cfg):
    print("⚠️ 已注册，不重复添加")
    sys.exit(0)

tmpl = None
for c in cfg:
    if (c.get("group") or "") == "安徽":
        tmpl = c
        break
if tmpl is None:
    tmpl = cfg[-1]
print("模板键集合:", sorted(tmpl.keys()))
print("模板样例:", json.dumps(tmpl, ensure_ascii=False)[:260])

entry = dict(tmpl)
entry.update({
    "name": "颍泉区人民政府-公告公示",
    "display_name": "颍泉区人民政府-公告公示",
    "script": SCRIPT,
    "args": ["--pages=5"],
    "group": "安徽",
    "note": "阜阳市颍泉区 公告公示, showList系(安徽站群) 12页224条/20条每页, "
            "第N页=/Content/showList/508/page_N.html(page_100+返回200空壳非404,靠0条停), "
            "列表 div.m-cglist>ul>li(日期span在a之前,侧栏他栏目同形态须限容器), "
            "正文 div#zoom, 附件区为#zoom兄弟节点 div.m-dtdownload(URL=/download?siteId=&id= 无扩展名)",
})
print("\n待追加:", json.dumps(entry, ensure_ascii=False)[:520])

cfg.append(entry)
os.makedirs(os.path.dirname(BAK), exist_ok=True)
if not os.path.exists(BAK):
    shutil.copy2(CFG, BAK)
with open(CFG, "w", encoding="utf-8") as f:
    json.dump(cfg, f, ensure_ascii=False, indent=2)

cfg2 = json.load(open(CFG, encoding="utf-8"))
print("\n注册后条目数:", len(cfg2))
missing = [c.get("script") for c in cfg2
           if c.get("script") and not os.path.exists(os.path.join("/root/gov_crawler", c["script"]))]
print("config 中 script 磁盘缺失:", len(missing), missing[:5])
idx = [i + 1 for i, c in enumerate(cfg2) if (c.get("script") or "") == SCRIPT]
print("✅ 已注册 %s  序号(1-based) = %s" % (SCRIPT, idx))
print("备份:", BAK)
