#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_jxth_api.py —— 服务器侧找泰和县环评栏目的真实 ajax 参数"""
import re, ssl, urllib.request
ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/141.0.0.0 Safari/537.36"}
B = "https://www.jxth.gov.cn"
def get(u):
    try:
        r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=20, context=ctx)
        return r.read(400000).decode("utf-8", "ignore")
    except Exception as e:
        return "ERR:" + type(e).__name__
h = get(B + "/")
print("首页:", len(h), "字节")
if not h.startswith("ERR"):
    ls = re.findall(r'href="([^"]+)"[^>]*>([^<]{0,40})', h)
    hits = [(u, t) for u, t in ls if re.search(r"环评|环境影响|受理|拟批", t)]
    print("环评相关链接 %d 条:" % len(hits))
    for u, t in hits[:8]:
        print("   %-56s %s" % (u[:56], t.strip()[:24]))
    # 页面内的栏目号线索
    for pat in (r"ajax_type[^;]{0,200}", r'catid["\']?\s*[:=]\s*["\']?(\d{3,})', r"(\d{6})"):
        g = re.findall(pat, h)
        if g: print("   线索 %-12s %s" % (pat[:12], str(g[:5])[:150]))
