#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""register_batch6.py —— 注册本批 5 个栏目（4 个脚本，缙云一脚本两栏目）"""
import json
import os
import shutil
import sys

CFG = "/root/gov_crawler/daily_crawl_config.json"
BAK = "/root/gov_crawler/Archive/daily_crawl_config.json.bak_20260924_batch6"

ENTRIES = [
    dict(name="宾阳县-建设项目环境影响评价审批", script="crawl_binyang_hjsp.py",
         group="广西", args=["--pages=5"],
         note="宾阳县 建设项目环评审批, TRS createPageHTML(19,0,index,html) 19页/20条每页, "
              "列表 li><a href=./tN.html title>…</a><span class=time>[日期]</span>, "
              "正文 div.trs_editor_view, 无 meta ArticleTitle(须剥<title>站点后缀), 附件 ./P020*.docx"),
    dict(name="大足区古龙镇-其他公文", script="crawl_dazu_glz_qtgw.py",
         group="重庆", args=["--pages=5"],
         note="重庆大足区古龙镇 其他公文, createPage(4,0,index,html) 4页57条/15条每页, "
              "列表 li><a href=./YYYYMM/tYYYYMMDD_ID.html>…</a><span>日期</span>, "
              "正文 div.zwxl-content > div.trs_editor_view(外层含标题副本+索引号元数据须取内层), "
              "短正文型公告(详情见附件/标题+附件图)属正常源站数据"),
    dict(name="青阳县-建设项目环境影响评价审批", script="crawl_ahqy_hjsp.py",
         group="安徽", args=["--pages=5"],
         note="青阳县 Jczwgk 基层政务公开, pagecount=30 / 20条每页, "
              "第N页=/Jczwgk/opennessList/650/102002008/page_N.html, "
              "列表 li><span>日期</span><a href=/Jczwgk/opennessShow/N.html title>, "
              "正文 div.m-dttexts(与颍泉同款 CMS)"),
    dict(name="缙云县-行政许可（受理）公示", script="crawl_jinyun_hjgs.py",
         group="浙江", args=["--col=1229425888", "--pages=5"],
         note="浙江 JPAAS API 站: GET /api-gateway/jpaas-publish-server/front/page/build/unit"
              "?webId=3658&tplSetId=PnEqYxUh1MkK3CjYn4xc5&tagId=列表&pageId=<col>&paramJson={pageNo,pageSize} "
              "→ JSON.data.html; count=431; 正文 div.content; 双URL格式(新/col/colNNN/art/ + 旧/art/Y/M/D/art_<colId>_N.html)"),
    dict(name="缙云县-建设项目环境影响评价信息公示", script="crawl_jinyun_hjgs.py",
         group="浙江", args=["--col=1229856959", "--pages=5"],
         note="同上(JPAAS API), count=86, 第2个栏目"),
]

cfg = json.load(open(CFG, encoding="utf-8"))
print("注册前条目数:", len(cfg))
tmpl = None
for c in cfg:
    if (c.get("group") or "") == "浙江":
        tmpl = c
        break
if tmpl is None:
    tmpl = cfg[-1]
print("模板键:", sorted(tmpl.keys()))

added = []
for e in ENTRIES:
    if any((c.get("script") or "") == e["script"] and (c.get("args") or []) == e["args"] for c in cfg):
        print("  ⚠️ 已存在，跳过:", e["name"])
        continue
    ent = dict(tmpl)
    ent.update({"name": e["name"], "display_name": e["name"], "script": e["script"],
                "args": e["args"], "group": e["group"], "note": e["note"],
                "incremental": True, "enabled": True, "sync_mode": "direct_server"})
    cfg.append(ent)
    added.append(e["name"])
print("新增条目:", len(added))
for a in added:
    print("   +", a)

os.makedirs(os.path.dirname(BAK), exist_ok=True)
if not os.path.exists(BAK):
    shutil.copy2(CFG, BAK)
with open(CFG, "w", encoding="utf-8") as f:
    json.dump(cfg, f, ensure_ascii=False, indent=2)

cfg2 = json.load(open(CFG, encoding="utf-8"))
print("\n注册后条目数:", len(cfg2))
missing = [c.get("script") for c in cfg2
           if c.get("script") and not os.path.exists(os.path.join("/root/gov_crawler", c["script"]))]
print("config 中 script 磁盘缺失:", len(missing), missing[:5])
for e in ENTRIES:
    idx = [i + 1 for i, c in enumerate(cfg2)
           if (c.get("script") or "") == e["script"] and (c.get("args") or []) == e["args"]]
    print("   %-40s 序号 %s" % (e["name"], idx))
print("备份:", BAK)
