#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""gen_batch6.py —— 从已验证模板派生 3 个新站脚本（宾阳/青阳/大足）

模板：宾阳/大足 ← crawl_danzhai_hjzf.py（TRS 静态）
      青阳      ← crawl_yingquan_gggs.py（安徽 Jczwgk）
每站替换：常量块 + list_page_url() + parse_list() + extract_body()
生成后：py_compile + 真实列表页冒烟（断言 parse_list 条数）
"""
import io
import os
import re

D = "/root/gov_crawler/"


def split_funcs(src):
    """按 \\n\\ndef 切顶层函数 → {name: text}"""
    chunks = src.split("\n\ndef ")
    head = chunks[0]
    funcs = {}
    order = []
    for c in chunks[1:]:
        nm = c.split("(")[0].strip()
        funcs[nm] = "def " + c
        order.append(nm)
    return head, funcs, order


def build(base_file, out_file, consts, list_page_url, parse_list, extract_body, region_block=None):
    src = io.open(D + base_file, encoding="utf-8").read()
    head, funcs, order = split_funcs(src)
    # 常量：替换 SITE_NAME/GROUP_NAME/SCRIPT_NAME/CATEGORY/BASE_URL/LIST_URL/COL_PATH
    for k, v in consts.items():
        src_head = head
        head = re.sub(r'(?m)^%s = .*$' % k, '%s = %s' % (k, v), head, count=1)
        if head == src_head:
            print("   ⚠️ %s 常量未替换" % k)
    # 文件头 docstring 的站点行（保持可读）
    if region_block:
        head = re.sub(r'(?s)^(\s*""".*?)(^"""\nimport)', region_block + r'\n\2', head, count=1, flags=re.M | re.S)
    funcs["list_page_url"] = list_page_url.strip()
    funcs["parse_list"] = parse_list.strip()
    funcs["extract_body"] = extract_body.strip()
    out = head.rstrip("\n") + "\n\n\n" + "\n\n\n".join(funcs[n] for n in order) + "\n"
    io.open(D + out_file, "w", encoding="utf-8").write(out)
    print("   生成 %s (%d 行)" % (out_file, out.count("\n") + 1))
    return out_file


# ── ① 宾阳 ────────────────────────────────────────────────────────────
CONST_BY = {
    "SITE_NAME": '"宾阳县-建设项目环境影响评价审批"',
    "GROUP_NAME": '"广西"',
    "SCRIPT_NAME": '"crawl_binyang_hjsp.py"',
    "CATEGORY": '"建设项目环境影响评价审批"',
    "BASE_URL": '"http://www.binyang.gov.cn"',
    "LIST_URL": '"http://www.binyang.gov.cn/gk/xxgkml/shgysyjslygk/hjbhly/jsxmhjyxpjsp/"',
    "COL_PATH": '"/jsxmhjyxpjsp/"',
}
LIS_BY = '''
def list_page_url(page):
    """createPageHTML(19,0,"index","html") → 第1页=目录, 第N页=index_(N-1).html"""
    if page <= 1:
        return LIST_URL
    return urllib.parse.urljoin(LIST_URL, "index_%d.html" % (page - 1))
'''
LIST_BY = r'''
def parse_list(html_text):
    """<li>…<a href="./tN.html" title="完整标题">…</a><span class="time">[YYYY-MM-DD]</span></li>"""
    items = []
    for m in re.finditer(r"<li[^>]*>(.*?)</li>", html_text, re.S | re.I):
        li = m.group(1)
        am = re.search(r'<a\b[^>]*href="([^"]+)"[^>]*title="([^"]*)"', li, re.I)
        if not am:
            am = re.search(r'<a\b[^>]*href="([^"]+)"[^>]*>(.*?)</a>', li, re.S | re.I)
        if not am:
            continue
        href = html_lib.unescape(am.group(1)).strip()
        if not re.search(r"/t\d+\.html$", href, re.I):
            continue
        title = clean_title(am.group(2) if am.lastindex and am.lastindex >= 2 else "")
        dm = re.search(r"\[?(\d{4})-(\d{1,2})-(\d{1,2})\]?", li)
        if not title or len(title) < 2 or not dm:
            continue
        date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
        items.append((urllib.parse.urljoin(LIST_URL, href), title, date))
    return items
'''
BODY_BY = '''
def extract_body(soup, html_text):
    """正文 = div.trs_editor_view.TRS_UEDITOR（TRS 富文本容器）"""
    el = soup.select_one("div.trs_editor_view") or soup.select_one("div[class*=TRS_UEDITOR]")
    if el is not None and el.get_text(strip=True):
        return "".join(str(c) for c in el.contents)
    el = soup.select_one("div.content") or soup.select_one("div.aticle_box")
    if el is not None:
        return "".join(str(c) for c in el.contents)
    return ""
'''

# ── ② 大足 ────────────────────────────────────────────────────────────
CONST_DZ = dict(CONST_BY)
CONST_DZ.update({
    "SITE_NAME": '"大足区古龙镇-其他公文"',
    "GROUP_NAME": '"重庆"',
    "SCRIPT_NAME": '"crawl_dazu_glz_qtgw.py"',
    "CATEGORY": '"其他公文"',
    "BASE_URL": '"https://www.dazu.gov.cn"',
    "LIST_URL": '"https://www.dazu.gov.cn/qzfjz/glz_101897/zwgk_53321/fdzdgknr_53323/lzyj_100471/qzfjz/"',
    "COL_PATH": '"/lzyj_100471/qzfjz/"',
})
LIST_DZ = r'''
def parse_list(html_text):
    """<li><a href="./YYYYMM/tYYYYMMDD_ID.html" title="完整标题">…</a><span>YYYY-MM-DD</span></li>"""
    items = []
    for m in re.finditer(r"<li[^>]*>(.*?)</li>", html_text, re.S | re.I):
        li = m.group(1)
        am = re.search(r'<a\b[^>]*href="([^"]+)"[^>]*title="([^"]*)"', li, re.I)
        if not am:
            continue
        href = html_lib.unescape(am.group(1)).strip()
        if not re.search(r"/t\d+_\d+\.html$", href, re.I):
            continue
        title = clean_title(am.group(2))
        dm = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", li)
        if not title or len(title) < 2 or not dm:
            continue
        date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
        items.append((urllib.parse.urljoin(LIST_URL, href), title, date))
    return items
'''
BODY_DZ = '''
def extract_body(soup, html_text):
    """正文 = div.zwxl-content > div.trs_editor_view（外层含标题副本与索引号元数据，取内层）"""
    el = soup.select_one("div.zwxl-content div.trs_editor_view")
    if el is None:
        el = soup.select_one("div.trs_editor_view")
    if el is not None and el.get_text(strip=True):
        return "".join(str(c) for c in el.contents)
    el = soup.select_one("div.zwxl-content")
    if el is not None:
        return "".join(str(c) for c in el.contents)
    return ""
'''

# ── ③ 青阳（基于颍泉模板，安徽 Jczwgk 同族）───────────────────────────
CONST_QY = {
    "SITE_NAME": '"青阳县-建设项目环境影响评价审批"',
    "GROUP_NAME": '"安徽"',
    "SCRIPT_NAME": '"crawl_ahqy_hjsp.py"',
    "CATEGORY": '"建设项目环境影响评价审批"',
    "BASE_URL": '"https://www.ahqy.gov.cn"',
    "LIST_URL": '"https://www.ahqy.gov.cn/Jczwgk/opennessList/650/102002008/page_1.html"',
    "COL_PATH": '"/Jczwgk/opennessShow/"',
}
LIS_QY = '''
def list_page_url(page):
    """pagecount=30；第 N 页 = /Jczwgk/opennessList/650/102002008/page_N.html"""
    return urllib.parse.urljoin(BASE_URL, "/Jczwgk/opennessList/650/102002008/page_%d.html" % page)
'''
LIST_QY = r'''
def parse_list(html_text):
    """<li><span>2026-09-21</span><a href="/Jczwgk/opennessShow/N.html" title="完整标题">…</a></li>"""
    items = []
    if not BeautifulSoup:
        return items
    soup = BeautifulSoup(html_text, "html.parser")
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = (a.get("href") or "").strip()
        if not href.startswith("/Jczwgk/opennessShow/") or not href.lower().endswith(".html"):
            continue
        span = li.find("span")
        date = ""
        if span:
            dm = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", span.get_text(" ", strip=True))
            if dm:
                date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
        if not date:
            continue
        title = clean_title(a.get("title") or a.get_text(" ", strip=True))
        if not title or len(title) < 2:
            continue
        items.append((urllib.parse.urljoin(BASE_URL, href), title, date))
    return items
'''
BODY_QY = '''
def extract_body(soup, html_text):
    """正文 = div.m-dttexts（含 j-fontContent；与颍泉同款 CMS）"""
    el = soup.select_one("div.m-dttexts") or soup.select_one("div.j-fontContent")
    if el is not None and el.get_text(strip=True):
        return "".join(str(c) for c in el.contents)
    el = soup.select_one("#zoom")
    if el is not None:
        return "".join(str(c) for c in el.contents)
    return ""
'''

print("=== 生成 ===")
build("crawl_danzhai_hjzf.py", "crawl_binyang_hjsp.py", CONST_BY, LIS_BY, LIST_BY, BODY_BY)
build("crawl_danzhai_hjzf.py", "crawl_dazu_glz_qtgw.py", CONST_DZ, LIS_BY, LIST_DZ, BODY_DZ)
build("crawl_yingquan_gggs.py", "crawl_ahqy_hjsp.py", CONST_QY, LIS_QY, LIST_QY, BODY_QY)

print("\n=== py_compile ===")
import py_compile
for f in ["crawl_binyang_hjsp.py", "crawl_dazu_glz_qtgw.py", "crawl_ahqy_hjsp.py"]:
    try:
        py_compile.compile(D + f, doraise=True)
        print("  ✅", f)
    except Exception as e:
        print("  ❌", f, str(e)[:120])
