#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""batch6_probe5.py —— 缙云API HTML结构 / 大足分页参数 / 江西厅栏目定性"""
import json
import re
import subprocess
import urllib.parse

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def sh(cmd):
    try:
        p = subprocess.run(cmd, shell=True, capture_output=True, timeout=60)
        return p.stdout.decode("utf-8", "ignore")
    except Exception as e:
        return "ERR " + str(e)[:70]


print("=" * 100)
print("【A. 缙云 API 返回的 HTML 结构】")
QD = {"webId": "3658", "tplSetId": "PnEqYxUh1MkK3CjYn4xc5", "tagId": "列表"}


def jy_api(col, page=1, size=20):
    pj = json.dumps({"pageNo": page, "pageSize": size}, ensure_ascii=False)
    url = ("https://www.jinyun.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
           "?parseType=bulidstatic&webId=%s&tplSetId=%s&pageType=column&tagId=%s&editType=null"
           "&pageId=%s&paramJson=%s" % (QD["webId"], QD["tplSetId"],
                                        urllib.parse.quote(QD["tagId"]), col, urllib.parse.quote(pj)))
    out = sh("curl -sk -L --max-time 40 -A '%s' '%s'" % (UA, url))
    try:
        return json.loads(out)["data"]["html"]
    except Exception:
        return ""


for col in ["1229425888", "1229856959"]:
    html = jy_api(col)
    print("\n  --- col%s API html (%d 字符) ---" % (col, len(html)))
    print("  片段:", re.sub(r"\s+", " ", html[:600]))
    lis = re.findall(r"<li[^>]*>(.*?)</li>", html, re.S | re.I)
    print("  li 数:", len(lis))
    for x in lis[:3]:
        print("    ", re.sub(r"\s+", " ", x)[:220])
    print("  所有 href:", list(dict.fromkeys(re.findall(r'href="([^"]+)"', html)))[:6])
    print("  日期:", sorted(set(re.findall(r"20\d{2}-\d{2}-\d{2}", html)))[:5])
    print("  分页线索:", list(dict.fromkeys(re.findall(r"count=\"?\d+|共\s*\d+\s*条|page_?\d+\.html", html)))[:6])
    # 翻页试探
    h2 = jy_api(col, page=2)
    d1 = sorted(set(re.findall(r"20\d{2}-\d{2}-\d{2}", html)))
    d2 = sorted(set(re.findall(r"20\d{2}-\d{2}-\d{2}", h2)))
    print("  page2 字节 %d | 与 page1 日期重叠: %d" % (len(h2), len(set(d1) & set(d2))))

print()
print("=" * 100)
print("【B. 大足分页参数（内联 JS 里的 createPageHTML 调用）】")
L = "https://www.dazu.gov.cn/qzfjz/glz_101897/zwgk_53321/fdzdgknr_53323/lzyj_100471/qzfjz/"
h = sh("curl -sk -L --max-time 40 -A '%s' '%s'" % (UA, L))
for pat in [r"createPageHTML\s*\(([^)]*)\)", r"_nPageCount\s*=\s*\d+", r"var\s+_nPageCount[^;]*;",
            r"总\d+条|共\s*\d+\s*条", r"pagedata[^\n]{0,120}", r"var\s+_sPageName[^;]*;"]:
    mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:4]
    if mm:
        print("  %-32s %s" % (pat[:32], [str(x)[:120] for x in mm]))
i = h.find("_nPageCount")
print("  上下文:", re.sub(r"\s+", " ", h[max(0, i - 200):i + 400])[:600] if i > 0 else "未找到")

print()
print("=" * 100)
print("【C. 江西厅 col42221 定性】")
for u in ["http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42221/index.html",
          "http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42221/index_1.html"]:
    h = sh("curl -sk -L --max-time 40 -A '%s' '%s'" % (UA, u))
    m = re.search(r"<title[^>]*>(.*?)</title>", h, re.S | re.I)
    print("  %s → %d 字节 | title: %s" % (u[-40:], len(h), m.group(1).strip()[:60] if m else "?"))
    qd = re.findall(r'queryData="([^"]{0,300})"', h)
    if qd:
        print("     queryData:", qd[0][:260])
    # 找表单/api
    for pat in [r"action=\"([^\"]+)\"", r"/api-[a-z-]+/[^\"'\s]+", r"iframe[^>]*src=\"([^\"]+)\"",
                r"\.jsp[^\"'\s]*", r"ajax[^\n]{0,80}"]:
        mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:3]
        if mm:
            print("     %-30s %s" % (pat[:30], [str(x)[:90] for x in mm]))
print()
print("  已覆盖核对：现有 3 个江西厅脚本覆盖哪些栏目？")
for f in ["crawl_jxsthjt_nslx.py", "crawl_jxsthjt_pzxmgg.py", "crawl_jxsthjt_npzgs.py"]:
    try:
        t = open("/root/gov_crawler/" + f, encoding="utf-8", errors="ignore").read()
    except Exception:
        continue
    print("   ", f, "| SITE_NAME:", re.findall(r'SITE_NAME\s*=\s*"([^"]+)"', t)[:1],
          "| col:", list(dict.fromkeys(re.findall(r"col(\d+)", t)))[:3],
          "| URL:", list(dict.fromkeys(re.findall(r"https?://[^\s\"']+col[^\s\"']+", t)))[:1])
