#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_yingquan.py —— 颍泉区 showList/508 栏目结构探测（含聚合页判据）"""
import re
import urllib.parse
from collections import Counter

import requests

requests.packages.urllib3.disable_warnings()
URL = "https://www.yingquan.gov.cn/Content/showList/508/page_1.html"
LIB = "https://www.yingquan.gov.cn"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept": "text/html,application/xhtml+xml,*/*;q=0.8",
     "Accept-Language": "zh-CN,zh;q=0.9"}

print("=" * 92)
r = requests.get(URL, headers=H, timeout=30, verify=False, allow_redirects=True)
r.encoding = r.apparent_encoding or "utf-8"
html = r.text
print("HTTP:", r.status_code, "| 最终URL:", r.url, "| 字节:", len(html))
m = re.search(r"<title[^>]*>(.*?)</title>", html, re.S | re.I)
print("<title>:", (m.group(1).strip()[:130] if m else "(无)"))

print("\n--- CMS 指纹 ---")
for k, pats in {
    "TRS": ["trs_editor", "TRS_Editor", "createPageHTML", "dynclicks"],
    "showList系": ["showList", "page_1.html"],
    "JPAAS": ["unitbuild", "paramJson", "api-gateway"],
    "Lonsun": ["label/8888", "site/label"],
    "Hanweb": ["dataproxy.jsp", "jpage"],
    "UCAP": ["UCAPCONTENT", "zoomcon"],
    "aspnet": ["__VIEWSTATE", ".aspx"],
    "php": [".php"],
}.items():
    hit = [p for p in pats if p in html]
    if hit:
        print("  %-10s → %s" % (k, hit))

print("\n--- 分页线索 ---")
for pat in [r"createPageHTML\([^)]*\)", r"page_\d+\.html", r"共\s*\d+\s*页", r"totalPage",
            r"pageCount", r"totalRecord", r"cur_page", r"pageIndex", r"showList/\d+/page_"]:
    mm = re.findall(pat, html, re.I)
    if mm:
        print("  %-28s → %s" % (pat[:28], list(dict.fromkeys(mm))[:6]))

print("\n--- 聚合页判据：条目链接的 netloc 分布 ---")
hrefs = re.findall(r'href=["\']([^"\']+)["\']', html)
nets = Counter()
for h in hrefs:
    if h.startswith(("javascript:", "#", "mailto:")):
        continue
    ab = urllib.parse.urljoin(URL, h)
    nets[urllib.parse.urlparse(ab).netloc] += 1
for n, c in nets.most_common(8):
    print("   %-32s %d" % (n, c))

print("\n--- 条目候选（含日期的 li/tr）---")
lis = re.findall(r"<li[^>]*>.*?</li>", html, re.S | re.I)
print("  <li>:", len(lis))
n = 0
for li in lis:
    if re.search(r"20\d{2}[-/.]\d{1,2}[-/.]\d{1,2}", li) and "<a" in li:
        n += 1
        if n <= 3:
            print("   ", re.sub(r"\s+", " ", li)[:300])
print("  含日期+链接的 li:", n)

print("\n--- 日期格式样例 ---")
print(" ", sorted(set(re.findall(r"20\d{2}[-/年.]\d{1,2}[-/月.]\d{1,2}日?", html)))[:10])

print("\n--- 容器 class 前 18 ---")
cls = re.findall(r'class=["\']([^"\']+)["\']', html)
for c, k in Counter([x.strip() for s in cls for x in s.split()]).most_common(18):
    print("   %-30s %d" % (c, k))

print("\n--- 分页块原文 ---")
i = html.find("page_")
if i < 0:
    i = html.find("分页")
print(" ", re.sub(r"\s+", " ", html[max(0, i - 400):i + 500])[:800] if i >= 0 else "  未找到")
