#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""gen_jinyun.py —— 生成 crawl_jinyun_hjgs.py（浙江 JPAAS API 站，双栏目）

列表走 JPAAS 接口（返回 JSON，data.html 里是列表片段）：
  GET /api-gateway/jpaas-publish-server/front/page/build/unit
      ?parseType=bulidstatic&webId=3658&tplSetId=PnEqYxUh1MkK3CjYn4xc5
      &pageType=column&tagId=列表&editType=null&pageId=<col>
      &paramJson={"pageNo":N,"pageSize":20}
条目：<li>…<a href="/col/colXXX/art/YYYY/art_HEX.html" title="完整标题">…</a>
      … <span class="zf_span4"> 2026-09-14 </span></li>
"""
import io
import os
import py_compile
import re

D = "/root/gov_crawler/"
src = io.open(D + "crawl_danzhai_hjzf.py", encoding="utf-8").read()

# 1) 常量块 → 双栏目字典 + --col 解析
c_old = re.search(r'(?m)^SITE_NAME = .*?\n^CUTOFF = .*?$', src, re.S).group(0)
c_new = '''COLUMNS = {
    "1229425888": ("缙云县-行政许可（受理）公示", "行政许可（受理）公示"),
    "1229856959": ("缙云县-建设项目环境影响评价信息公示", "建设项目环境影响评价信息公示"),
}
_col = ""
for i, a in enumerate(sys.argv):
    if a.startswith("--col="):
        _col = a.split("=", 1)[1].strip()
    elif a == "--col" and i + 1 < len(sys.argv):
        _col = sys.argv[i + 1].strip()
if not _col or _col not in COLUMNS:
    print("用法: python3 crawl_jinyun_hjgs.py --col=<%s> [--pages=5]" % "|".join(COLUMNS))
    sys.exit(1)
SITE_NAME, CATEGORY = COLUMNS[_col]
GROUP_NAME = "浙江"
SCRIPT_NAME = "crawl_jinyun_hjgs.py"
BASE_URL = "https://www.jinyun.gov.cn"
LIST_URL = "https://www.jinyun.gov.cn/col/col%s/index.html" % _col
COL_PATH = "/col/col%s/art/" % _col
JY_API = ("https://www.jinyun.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
          "?parseType=bulidstatic&webId=3658&tplSetId=PnEqYxUh1MkK3CjYn4xc5"
          "&pageType=column&tagId=%s&editType=null&pageId=%s&paramJson=%s")
JY_TAG = urllib.parse.quote("列表")
CUTOFF = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")'''
assert src.count(c_old) == 1
src = src.replace(c_old, c_new)


def split_funcs(s):
    chunks = s.split("\n\ndef ")
    head = chunks[0]
    funcs, order = {}, []
    for c in chunks[1:]:
        nm = c.split("(")[0].strip()
        funcs[nm] = "def " + c
        order.append(nm)
    return head, funcs, order


head, funcs, order = split_funcs(src)

funcs["list_page_url"] = '''def list_page_url(page):
    """列表走 JPAAS 接口；第 N 页 = paramJson.pageNo=N"""
    pj = json.dumps({"pageNo": page, "pageSize": 20}, ensure_ascii=False)
    return JY_API % (JY_TAG, _col, urllib.parse.quote(pj))'''

funcs["parse_list"] = r'''def parse_list(api_json_text):
    """接口返回 {"success":true,"data":{"html":"<div id=列表>…</div>"}}，从 data.html 解析条目"""
    items = []
    try:
        d = json.loads(api_json_text)
        html_text = (d.get("data") or {}).get("html") or ""
    except Exception:
        return items
    for m in re.finditer(r"<li[^>]*>(.*?)</li>", html_text, re.S | re.I):
        li = m.group(1)
        am = re.search(r'<a\b[^>]*href="([^"]+)"[^>]*title="([^"]*)"', li, re.I)
        if not am:
            continue
        href = html_lib.unescape(am.group(1)).strip()
        if COL_PATH not in href or not href.lower().endswith(".html"):
            continue
        title = clean_title(am.group(2))
        if not title or len(title) < 2:
            continue
        dm = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", li)
        if not dm:
            continue
        date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
        items.append((urllib.parse.urljoin(BASE_URL, href), title, date))
    return items'''

funcs["extract_body"] = '''def extract_body(soup, html_text):
    """浙江 JPAAS 正文容器（按参考：div.content > div.wenz 优先，逐级回退）"""
    for sel in ("div.content div.wenz", "div.wenz", "#zoom", "div.zwxl-content",
                "div.trs_editor_view", "div.article", "div.Article_Con"):
        el = soup.select_one(sel)
        if el is not None and el.get_text(strip=True):
            return "".join(str(c) for c in el.contents)
    return ""'''

# main(): 列表 fetch 是 JSON，不要当 HTML 解析；其余沿用
funcs["main"] = funcs["main"].replace(
    '        items = parse_list(html_text)',
    '        items = parse_list(html_text)  # html_text 此处为接口 JSON')

out = head.rstrip("\n") + "\n\n\n" + "\n\n\n".join(funcs[n] for n in order) + "\n"
# 清掉 danzhai 遗留的默认 SCRIPT_NAME 打印等
io.open(D + "crawl_jinyun_hjgs.py", "w", encoding="utf-8").write(out)
print("生成 crawl_jinyun_hjgs.py (%d 行)" % (out.count("\n") + 1))
try:
    py_compile.compile(D + "crawl_jinyun_hjgs.py", doraise=True)
    print("✅ py_compile 通过")
except Exception as e:
    print("❌ py_compile:", str(e)[:200])
