#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_qdndz.py —— 青岛某站 环境执法结果 栏目结构探测"""
import re
import urllib.parse

import requests

requests.packages.urllib3.disable_warnings()
URL = "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept": "text/html,application/xhtml+xml,*/*;q=0.8",
     "Accept-Language": "zh-CN,zh;q=0.9"}

print("=" * 90)
print("列表页:", URL)
try:
    r = requests.get(URL, headers=H, timeout=30, verify=False, allow_redirects=True)
    print("  HTTP:", r.status_code, "| 最终URL:", r.url, "| 大小:", len(r.content))
    r.encoding = r.apparent_encoding or "utf-8"
    html = r.text
except Exception as e:
    print("  ❌ 失败:", e)
    raise SystemExit(1)

m = re.search(r"<title[^>]*>(.*?)</title>", html, re.S | re.I)
print("  <title>:", (m.group(1).strip()[:120] if m else "(无)"))

print("\n--- CMS 指纹 ---")
prints = {
    "TRS(拓尔思)": ["trs_editor", "TRS_Editor", "createPageHTML", "dynclicks", "/protect/", "was5"],
    "JPAAS(浙江)": ["unitbuild", "paramJson", "jpaas", "api-gateway"],
    "Lonsun(龙讯)": ["label/8888", "site/label", "xxgk_nav"],
    "Hanweb(大汉)": ["dataproxy.jsp", "jpage", "truecms"],
    "nfw-cms": ["nfw-cms-attachment", "article-content"],
    "UCAP": ["UCAPCONTENT", "zoomcon", "ucap"],
    "jsp/asp/php": [".jsp", ".aspx", ".php"],
    "es-search": ["es-search", "lmCode", "channelId"],
}
for k, pats in prints.items():
    hit = [p for p in pats if p in html]
    if hit:
        print("  %-12s → %s" % (k, hit))

print("\n--- 分页线索 ---")
for pat in [r"createPageHTML\([^)]*\)", r"totalPage\s*[:=]\s*\d+", r"totalpage\s*[:=]\s*\d+",
            r"pageCount\s*[:=]\s*\d+", r"totalRecord\s*[:=]\s*\d+", r"index_\d+\.html",
            r"共\s*(\d+)\s*页", r"cur_page", r"pageIndex", r"nr_page", r"\.jpage", r"startrecord"]:
    mm = re.findall(pat, html, re.I)
    if mm:
        print("  %-24s → %s" % (pat[:24], mm[:6]))

print("\n--- 列表条目候选（前 8 条）---")
# a[title]
items = re.findall(r'<a[^>]+href=["\']([^"\']+)["\'][^>]*title=["\']([^"\']{4,120})["\']', html, re.I)
print("  带 title 的 a 标签:", len(items), "个")
for u, t in items[:8]:
    print("    ", t[:60], "|", u[:80])

print("\n--- 日期格式样例 ---")
dates = re.findall(r"(20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}日?)", html)
print("  ", sorted(set(dates))[:12])

print("\n--- 容器 class 频次（前 20）---")
cls = re.findall(r'class=["\']([^"\']+)["\']', html)
from collections import Counter
for c, n in Counter([x.strip() for s in cls for x in s.split()]).most_common(20):
    print("  %-30s %d" % (c, n))

print("\n--- 是否含正文容器关键词 ---")
for k in ["TRS_Editor", "trs_editor_view", "article", "content", "zoom", "wzcon", "detail", "conTxt", "view"]:
    print("  %-18s %d 次" % (k, html.count(k)))

# 详情页探测
cands = [u for u, t in items if not u.startswith(("javascript:", "#", "mailto:"))]
if cands:
    d = urllib.parse.urljoin(URL, cands[0])
    print("\n" + "=" * 90)
    print("详情页:", d)
    try:
        r2 = requests.get(d, headers=H, timeout=30, verify=False)
        r2.encoding = r2.apparent_encoding or "utf-8"
        h2 = r2.text
        print("  HTTP:", r2.status_code, "| 大小:", len(h2))
        m2 = re.search(r"<title[^>]*>(.*?)</title>", h2, re.S | re.I)
        print("  <title>:", (m2.group(1).strip()[:120] if m2 else "(无)"))
        for pat in [r'<meta[^>]+name=["\']?(?:ArticleTitle|PubDate|ColumnName|ContentSource)["\']?[^>]*>',
                    r'<h1[^>]*>(.*?)</h1>', r'<h2[^>]*>(.*?)</h2>']:
            mm = re.findall(pat, h2, re.S | re.I)
            if mm:
                print("  %s → %s" % (pat[:34], [str(x)[:70] for x in mm[:3]]))
        cls2 = re.findall(r'class=["\']([^"\']+)["\']', h2)
        print("  容器 class 前 15:")
        for c, n in Counter([x.strip() for s in cls2 for x in s.split()]).most_common(15):
            print("    %-28s %d" % (c, n))
        atts = re.findall(r'<a[^>]+href=["\']([^"\']+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|jpg|png))["\']', h2, re.I)
        print("  附件/文件链接:", len(atts), atts[:5])
    except Exception as e:
        print("  ❌ 详情失败:", e)
else:
    print("\n  ⚠️ 列表未提取到可用详情链接")
