#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_qdndz3.py —— 页面 URL 形态 + ArchiveGdPart 是否附件区 + 正文/附件样例"""
import re
import urllib.parse

import requests

requests.packages.urllib3.disable_warnings()
BASE = "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept": "text/html,application/xhtml+xml,*/*;q=0.8",
     "Accept-Language": "zh-CN,zh;q=0.9"}


def get(u):
    r = requests.get(u, headers=H, timeout=30, verify=False)
    r.encoding = r.apparent_encoding or "utf-8"
    return r.status_code, r.text


print("=== ① 分页 URL 形态实测 ===")
for u in [BASE, BASE + "index.html", BASE + "index_1.html", BASE + "index_2.html",
          BASE + "index_1.shtml", BASE + "index_18.html", BASE + "index_19.html"]:
    try:
        code, h = get(u)
        n = len(re.findall(r'<ul class="NewsList">.*?</ul>', h, re.S)[0].split("<li")) - 1 if 'class="NewsList"' in h else -1
        first = ""
        m = re.search(r'<ul class="NewsList">.*?title="([^"]{4,80})".*?<span>(20\d\d-\d\d-\d\d)', h, re.S)
        if m:
            first = m.group(2) + " " + m.group(1)[:34]
        print("  %-72s HTTP %s 条目=%s 首条=%s" % (u.replace(BASE, "(base)"), code, n, first))
    except Exception as e:
        print("  %-72s ❌ %s" % (u.replace(BASE, "(base)"), str(e)[:40]))

print("\n=== ② 详情页正文/附件区结构 ===")
det = ["https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/202608/t20260803_90687190.html",
       "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/202606/t20260623_90547406.html"]
for d in det:
    code, h = get(d)
    print("\n---- %s (HTTP %s, %d 字节) ----" % (d[-32:], code, len(h)))
    for name in ["Article_Con", "ArchiveGdPart", "ArticleProperties"]:
        i = h.find(name)
        if i < 0:
            print("  [%s] 未出现" % name)
            continue
        s = h.rfind("<div", 0, i)
        frag = h[s:s + 1400]
        print("  [%s] 片段:" % name)
        print("   ", re.sub(r"\s+", " ", frag)[:800])
        print("    → 含 <a href> :", len(re.findall(r"<a[^>]+href", frag)),
              "| 含文件后缀:", re.findall(r'[\w\-\.]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|jpg|png)', frag)[:4])
    # 全页附件
    files = re.findall(r'<a[^>]+href=["\']([^"\']+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))["\']', h, re.I)
    print("  全页文件链接:", len(files), files[:4])
    print("  正文 p 数:", h.count("<p"), "| table 数:", h.count("<table"))
    # 图片（可能是扫描件）
    imgs = re.findall(r'<img[^>]+src=["\']([^"\']+)["\']', h)
    print("  img 数:", len(imgs), [x[-40:] for x in imgs[:4]])
