#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_qdndz2.py —— 看清列表条目真实标记 + 分页块 + 详情页结构"""
import re
import urllib.parse
from collections import Counter

import requests

requests.packages.urllib3.disable_warnings()
URL = "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept": "text/html,application/xhtml+xml,*/*;q=0.8",
     "Accept-Language": "zh-CN,zh;q=0.9"}

r = requests.get(URL, headers=H, timeout=30, verify=False)
r.encoding = r.apparent_encoding or "utf-8"
html = r.text

print("=== ① 日期出现处上下文（第 1 个日期前后 900 字符）===")
m = re.search(r"20\d{2}-\d{2}-\d{2}", html)
if m:
    s = max(0, m.start() - 700)
    print(html[s:m.end() + 200])
else:
    print("  ⚠️ 未找到日期")

print("\n\n=== ② 所有 li 中含日期的条目 ===")
lis = re.findall(r"<li[^>]*>.*?</li>", html, re.S | re.I)
print("  总 li 数:", len(lis))
n = 0
for li in lis:
    if re.search(r"20\d{2}-\d{2}-\d{2}", li) and "<a" in li:
        n += 1
        if n <= 6:
            print("  ── 第 %d 条 ──" % n)
            print("   ", re.sub(r"\s+", " ", li)[:300])
print("  含日期+链接的 li:", n, "个")

print("\n\n=== ③ 候选详情 URL 模式 ===")
hrefs = re.findall(r'href=["\']([^"\']+)["\']', html)
pats = Counter()
for h in hrefs:
    if h.startswith(("javascript:", "#", "mailto:", "http://www.w3.org")):
        continue
    pats[re.sub(r"\d+", "N", h)[:60]] += 1
for p, c in pats.most_common(18):
    print("  %-62s %d" % (p, c))

print("\n\n=== ④ 分页块原文 ===")
for kw in ["createPageHTML", "page_div", "分页", "下一页", "CurrChnlCls"]:
    i = html.find(kw)
    if i >= 0:
        print("  【%s】..." % kw)
        print("   ", re.sub(r"\s+", " ", html[max(0, i - 300):i + 400])[:600])
        break

print("\n\n=== ⑤ 挑一个真·详情页探测 ===")
cand = [h for h in hrefs if re.search(r"/\d{6}/t\d+.*\.s?html?", h) or re.search(r"/20\d{2}\d{2}/", h)]
cand += [h for h in hrefs if re.search(r"\.s?html", h) and "search.shtml" not in h and "/so/" not in h]
seen = []
for h in cand:
    if h not in seen:
        seen.append(h)
print("  候选:", seen[:6])
if seen:
    d = urllib.parse.urljoin(URL, seen[0])
    print("  实测详情:", d)
    r2 = requests.get(d, headers=H, timeout=30, verify=False)
    r2.encoding = r2.apparent_encoding or "utf-8"
    h2 = r2.text
    print("  HTTP:", r2.status_code, "| 大小:", len(h2))
    m2 = re.search(r"<title[^>]*>(.*?)</title>", h2, re.S | re.I)
    print("  <title>:", (m2.group(1).strip()[:100] if m2 else "(无)"))
    for pat in [r'<meta[^>]*name=["\']?(ArticleTitle|PubDate|ColumnName|ContentSource|Author)["\']?[^>]*>']:
        mm = re.findall(pat, h2, re.I)
        if mm:
            print("  meta:", mm[:6])
    # 正文容器：找 <p> 最多的 div
    print("  各 class 的 <p> 数（前 12）:")
    blocks = re.findall(r'<div[^>]*class=["\']([^"\']+)["\'][^>]*>(.*?)</div>\s*(?=<div|</body|$)', h2, re.S | re.I)
    stats = []
    for cls, body in blocks:
        stats.append((body.count("<p"), len(re.sub(r"<[^>]+>", "", body)), cls[:40]))
    for pc, txt, cls in sorted(stats, reverse=True)[:12]:
        print("    p=%-3d 文本=%-6d %s" % (pc, txt, cls))
    atts = re.findall(r'<a[^>]+href=["\']([^"\']+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))["\']', h2, re.I)
    print("  附件:", len(atts), atts[:5])
