#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_jxst42221.py —— 江西厅 col42221 到底是不是列表页（深挖）"""
import re
import subprocess

from bs4 import BeautifulSoup

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def get(u):
    p = subprocess.run(["curl", "-sk", "-L", "--max-time", "40", "-A", UA,
                        "-H", "Accept-Language: zh-CN,zh;q=0.9", u],
                       capture_output=True, timeout=60)
    return p.stdout.decode("utf-8", "ignore")


TARGET = "http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42221/index.html"
SIB = "http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42169/index.html"   # 已有脚本覆盖的栏目

for tag, u in [("col42221（用户给的）", TARGET), ("col42169（已有脚本）", SIB)]:
    h = get(u)
    print("=" * 100)
    print("【%s】%s  →  %d 字节" % (tag, u[-38:], len(h)))
    if not h:
        print("  ❌ 空")
        continue
    soup = BeautifulSoup(h, "html.parser")
    m = re.search(r"<title[^>]*>(.*?)</title>", h, re.S | re.I)
    print("  title:", m.group(1).strip()[:80] if m else "?")
    # 正文文字（去掉 script/style）
    body = BeautifulSoup(re.sub(r"(?is)<(script|style)[^>]*>.*?</\1>", "", h), "html.parser")
    txt = re.sub(r"\s+", " ", body.get_text(" ", strip=True))
    print("  可见文本(%d字)前 320:" % len(txt), txt[:320])
    # 各种线索
    print("  li 数:", len(h.split("<li")), "| 含日期链接数:",
          len([x for x in re.findall(r"<li[^>]*>.*?</li>", h, re.S | re.I)
               if re.search(r"20\d{2}[-/.]\d{1,2}[-/.]\d{1,2}", x) and "<a" in x]))
    for pat in [r'queryData="([^"]{0,260})"', r"createPageHTML\([^)]*\)", r"pagecount[^,;>]{0,20}",
                r"totalpage[^,;>]{0,20}", r"col\d+/(?:index|list)[^\"'\s]{0,30}",
                r'<iframe[^>]*src="([^"]+)"', r"\.js\?v?=[^\"']{0,30}",
                r"/api-[a-z-]+/[^\"'\s]{0,60}", r"paramJson", r"unitbuild"]:
        mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:3]
        if mm:
            print("  %-30s %s" % (pat[:30], [str(x)[:110] for x in mm]))
    # 所有 a 的 href 归类
    hrefs = [a.get("href", "") for a in soup.find_all("a", href=True)]
    from collections import Counter
    c = Counter(re.sub(r"\d+", "N", x)[:52] for x in hrefs if not x.startswith(("javascript:", "#")))
    print("  a[href] 形态 top8:", c.most_common(8))
    # 带 title 属性的 a（列表特征）
    titled = [(a.get("href", ""), a.get("title", "")) for a in soup.find_all("a", title=True)]
    print("  带 title 的 a: %d" % len(titled))
    for u2, t in titled[:4]:
        print("     %s | %s" % (t[:52], u2[:64]))
    # script 标签
    srcs = [s.get("src") for s in soup.find_all("script", src=True)]
    print("  script src:", [s for s in srcs if s][:6])
    # 是否有 XHR / fetch / ajax 调用
    for kw in ["XMLHttpRequest", "fetch(", "$.ajax", "$.get", "$.post", "axios"]:
        if kw in h:
            i = h.find(kw)
            print("  ⭐ 发现 %s → %s" % (kw, re.sub(r"\s+", " ", h[max(0, i - 120):i + 180])[:280]))
