#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""batch6_probe4.py —— 缙云API实测 / 大足正文精确容器+分页 / 江西厅http列表与详情"""
import json
import re
import subprocess
import urllib.parse

from bs4 import BeautifulSoup

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def sh(cmd):
    try:
        p = subprocess.run(cmd, shell=True, capture_output=True, timeout=60)
        return p.stdout.decode("utf-8", "ignore")
    except Exception as e:
        return "ERR " + str(e)[:70]


def get(url):
    return sh("curl -sk -L --max-time 40 -A '%s' '%s'" % (UA, url))


print("=" * 100)
print("【A. 缙云 JPAAS API 实测】")
QD = {"parseType": "bulidstatic", "webId": "3658", "tplSetId": "PnEqYxUh1MkK3CjYn4xc5",
      "pageType": "column", "tagId": "列表", "editType": "null"}
for col in ["1229425888", "1229856959"]:
    for size in [20]:
        pj = json.dumps({"pageNo": 1, "pageSize": size}, ensure_ascii=False)
        url = ("https://www.jinyun.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
               "?parseType=%s&webId=%s&tplSetId=%s&pageType=column&tagId=%s&editType=null"
               "&pageId=%s&paramJson=%s" % (QD["parseType"], QD["webId"], QD["tplSetId"],
                                            urllib.parse.quote(QD["tagId"]), col,
                                            urllib.parse.quote(pj)))
        out = get(url)
        print("\n  col%s size=%d → %d 字节" % (col, size, len(out)))
        if not out:
            continue
        print("    开头:", out[:200].replace("\n", " "))
        # 找条目
        items = re.findall(r'<a[^>]+href="([^"]+)"[^>]*title="([^"]*)"', out)
        print("    带 title 的 a: %d" % len(items))
        for u, t in items[:5]:
            print("       ", t[:52], "|", u[:70])
        dates = re.findall(r"20\d{2}-\d{2}-\d{2}", out)
        print("    日期样例:", sorted(set(dates))[:5])
        cnt = re.findall(r'count="(\d+)"|共\s*(\d+)\s*条|totalCount["\']?\s*[:=]\s*"?(\d+)', out)
        print("    总数线索:", cnt[:3])

print()
print("=" * 100)
print("【B. 大足：正文精确容器 + 分页 URL】")
L = "https://www.dazu.gov.cn/qzfjz/glz_101897/zwgk_53321/fdzdgknr_53323/lzyj_100471/qzfjz/"
h = get(L)
print("  列表字节:", len(h))
for pat in [r'href="([^"]*index[_0-9]*\.html)"', r'class="(?:first|last|prev|next)-page"[^>]*>', r'href="([^"]*page[^"]*)"']:
    mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:6]
    if mm:
        print("  %-30s %s" % (pat[:30], [str(x)[:60] for x in mm]))
i = h.find("last-page")
if i > 0:
    print("  分页块:", re.sub(r"\s+", " ", h[max(0, i - 500):i + 300])[:600])
d = "https://www.dazu.gov.cn/qzfjz/glz_101897/zwgk_53321/fdzdgknr_53323/lzyj_100471/qzfjz/202609/t20260916_16085275.html"
hd = get(d)
soup = BeautifulSoup(hd, "html.parser")
print("\n  详情字节:", len(hd))
print("  候选容器（排除含 索引号/当前位置 的）:")
cands = []
for dv in soup.find_all("div"):
    tx = dv.get_text(" ", strip=True)
    if len(tx) < 100 or "索引号" in tx[:400] or "当前位置" in tx[:200]:
        continue
    inner = "".join(str(c) for c in dv.contents)
    cls = " ".join(dv.get("class") or []) or (dv.get("id") or "")
    cands.append((inner.count("<p"), len(tx), cls))
for pc, tx, cls in sorted(cands, reverse=True)[:8]:
    print("     p=%-3d 文本=%-6d %s" % (pc, tx, cls[:44]))
best = max(cands)[2] if cands else None
if best:
    dv = soup.find("div", class_=best.split()[0]) if best else None
    if dv:
        print("     选中:", best, "| 文本:", dv.get_text(" ", strip=True)[:200])
print("  含 table:", hd.count("<table"), "| 含 trs_editor:", "trs_editor" in hd.lower())

print()
print("=" * 100)
print("【C. 江西厅 http 列表 + 详情】")
LU = "http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42221/index.html"
h = get(LU)
print("  列表字节:", len(h))
m = re.search(r"<title[^>]*>(.*?)</title>", h, re.S | re.I)
print("  title:", m.group(1).strip()[:90] if m else "(无)")
print("  指纹:", {k: [p for p in v if p in h] for k, v in {
    "TRS": ["trs_editor", "createPageHTML", "dynclicks"],
    "jcms/JPAAS": ["col/col", "unitbuild", "jpaas"],
    "vue": ["__NUXT__"],
}.items() if any(p in h for p in v)})
for pat in [r"createPageHTML\([^)]*\)", r"page_\d+\.html", r"index_\d+\.html",
            r'queryData="([^"]{0,200})"', r"pagecount[\"'\s:=]+(\d+)"]:
    mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:3]
    if mm:
        print("  %-28s %s" % (pat[:28], [str(x)[:120] for x in mm]))
lis = re.findall(r"<li[^>]*>.*?</li>", h, re.S | re.I)
dated = [x for x in lis if re.search(r"20\d{2}[-/.]\d{1,2}", x) and "<a" in x]
print("  li %d | 含日期 %d" % (len(lis), len(dated)))
for x in dated[:3]:
    print("    ", re.sub(r"\s+", " ", x)[:230])
