#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""batch6_probe2.py —— 深挖：江西厅重试 / 缙云JPAAS API / 三站详情结构 / 已覆盖核对"""
import re
import subprocess
import urllib.parse
import json

import requests
from bs4 import BeautifulSoup

requests.packages.urllib3.disable_warnings()
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"}


def curl(url, extra=None):
    cmd = ["curl", "-sk", "-L", "--max-time", "30", "-A", UA] + (extra or []) + [url]
    try:
        p = subprocess.run(cmd, capture_output=True, timeout=45)
        return p.stdout.decode("utf-8", "ignore")
    except Exception as e:
        return ""


print("=" * 100)
print("【A. 江西生态环境厅 col42221：curl -sk 重试】")
u = "https://sthjt.jiangxi.gov.cn/jxssthjt/col/col42221/index.html"
h = curl(u)
print("  curl 字节:", len(h))
if h:
    m = re.search(r"<title[^>]*>(.*?)</title>", h, re.S | re.I)
    print("  title:", m.group(1).strip()[:90] if m else "(无)")
    print("  指纹:", {k: [p for p in v if p in h] for k, v in {
        "TRS": ["trs_editor", "createPageHTML", "dynclicks"],
        "jcms": ["col/col", "jcms", "unitbuild"],
        "vue": ["__NUXT__", "window.__INITIAL"],
    }.items() if any(p in h for p in v)})
    lis = re.findall(r"<li[^>]*>.*?</li>", h, re.S | re.I)
    dated = [x for x in lis if re.search(r"20\d{2}[-/.]\d{1,2}", x) and "<a" in x]
    print("  li %d | 含日期 %d" % (len(lis), len(dated)))
    for x in dated[:2]:
        print("    %s" % re.sub(r"\s+", " ", x)[:240])
    for pat in [r"createPageHTML\([^)]*\)", r"page_\d+\.html", r"index_\d+\.html", r"col\d+"]:
        mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:4]
        if mm:
            print("  分页 %-24s %s" % (pat[:24], mm))

print()
print("=" * 100)
print("【B. 缙云 JPAAS：找 API 与真实列表】")
for col in ["1229425888", "1229856959"]:
    u = "https://www.jinyun.gov.cn/col/col%s/index.html" % col
    h = curl(u)
    print("\n  --- col%s（%d 字节）---" % (col, len(h)))
    # 页面里的关键 JS 变量
    for pat in [r'jsearch[^"\']*', r'colId["\']?\s*[:=]\s*["\']?(\d+)', r'unitbuild[^"\']*',
                r'paramJson[^"\']*', r'webId["\']?\s*[:=]\s*["\']?(\d+)', r'tagId["\']?\s*[:=]\s*["\']?([^"\',}]+)',
                r'/module/jpage[^"\']*', r'"/api/[^"]+"', r'ajax[^"\']{0,60}']:
        mm = list(dict.fromkeys(re.findall(pat, h, re.I)))[:3]
        if mm:
            print("     %-30s %s" % (pat[:30], [str(x)[:80] for x in mm]))
    # script src
    srcs = re.findall(r'<script[^>]+src=["\']([^"\']+)["\']', h, re.I)
    print("     scripts:", [s for s in srcs if "jsearch" in s or "unitbuild" in s or "col" in s][:5])
    # 详情链接形态
    hrefs = re.findall(r'href=["\']([^"\']+)["\']', h)
    det = [x for x in hrefs if re.search(r"/art/|/art_|\.html", x) and "col% s" % "" not in x][:6]
    print("     可能详情链接:", det[:5])

print()
print("=" * 100)
print("【C. 三站详情页结构】")
DET = [
    ("宾阳", "http://www.binyang.gov.cn/gk/xxgkml/shgysyjslygk/hjbhly/jsxmhjyxpjsp/", r'href="(\./t\d+\.html)"'),
    ("青阳", "https://www.ahqy.gov.cn/Jczwgk/opennessList/650/102002008/page_1.html", r'href="(/Jczwgk/opennessShow/\d+\.html)"'),
    ("大足", "https://www.dazu.gov.cn/qzfjz/glz_101897/zwgk_53321/fdzdgknr_53323/lzyj_100471/qzfjz/", r'href="(\./\d+/t\d+\.html)"'),
]
for tag, listurl, pat in DET:
    h = curl(listurl)
    m = re.search(pat, h)
    if not m:
        print("\n  【%s】未找到详情链接" % tag)
        continue
    d = urllib.parse.urljoin(listurl, m.group(1))
    hd = curl(d)
    print("\n  【%s】%s (%d 字节)" % (tag, d[:78], len(hd)))
    mt = re.search(r"<title[^>]*>(.*?)</title>", hd, re.S | re.I)
    print("     title: %s" % (mt.group(1).strip()[:80] if mt else "(无)"))
    soup = BeautifulSoup(hd, "html.parser")
    for k in ["ArticleTitle", "PubDate", "ColumnName", "ContentSource"]:
        mm = soup.find("meta", attrs={"name": re.compile(k, re.I)})
        if mm and mm.get("content"):
            print("     meta %-14s %s" % (k, mm["content"][:60]))
    # 容器：<p> 最多的 div
    stats = []
    for dv in soup.find_all("div"):
        inner = "".join(str(c) for c in dv.contents)
        if dv.find("div") is not None:
            continue
        stats.append((inner.count("<p"), len(dv.get_text(" ", strip=True)), (dv.get("class") or [""])[0] if dv.get("class") else (dv.get("id") or "")))
    for pc, tx, cls in sorted(stats, reverse=True)[:5]:
        print("     容器 p=%-3d 文本=%-6d %s" % (pc, tx, cls))
    atts = re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd))"', hd, re.I)
    print("     附件(按扩展名): %d %s" % (len(atts), atts[:3]))
