#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_qdndz4.py —— 探环评公示类文章的附件形态（报批前公示通常带 PDF 全文）"""
import re

import requests

requests.packages.urllib3.disable_warnings()
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Referer": "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/"}

DETS = [
    ("报批前公示", "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/202609/t20260920_90920369.html"),
    ("第二次公示", "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/202608/t20260817_90738216.html"),
    ("公众参与第一次", "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/202609/t20260922_90929865.html"),
]
for tag, d in DETS:
    r = requests.get(d, headers=H, timeout=30, verify=False)
    r.encoding = r.apparent_encoding or "utf-8"
    h = r.text
    # 正文片段
    i = h.find('class="Article_Con')
    s = h.rfind("<div", 0, i)
    body = h[s:s + 20000]
    body = body[:body.find('</font>')] if '</font>' in body else body
    tx = re.sub(r"<[^>]+>", "", body)
    tx = re.sub(r"\s+", " ", tx).strip()
    print("=" * 88)
    print("【%s】HTTP %s 字节=%d" % (tag, r.status_code, len(h)))
    print("  正文纯文本长度:", len(tx))
    print("  正文前 200 字:", tx[:200])
    print("  正文后 200 字:", tx[-200:])
    print("  正文内 <p>:", body.count("<p"), "|<table>:", body.count("<table"), "|<img>:", body.count("<img"))
    # 全页所有 a[href]
    links = re.findall(r'<a[^>]+href=["\']([^"\']+)["\'][^>]*>(.*?)</a>', h, re.S | re.I)
    files, mail, other = [], [], []
    for u, t in links:
        t2 = re.sub(r"<[^>]+>", "", t).strip()[:40]
        if re.search(r"\.(?:pdf|doc|docx|xls|xlsx|zip|rar|jpg|png)", u, re.I):
            files.append((u, t2))
        elif u.startswith(("mailto:", "tel:", "javascript:", "#")):
            mail.append((u[:28], t2))
        elif "Article_Con" in h and t2 and len(t2) > 2 and re.search(r"download|file|attach|upload", u, re.I):
            other.append((u, t2))
    print("  文件类链接:", len(files))
    for u, t in files[:6]:
        print("    →", t, "|", u[:110])
    print("  mailto/tel/# 类:", len(mail), mail[:4])
    print("  含 download/file 关键词的链接:", len(other), other[:3])
    # 图片懒加载
    lo = re.findall(r'data-original=["\']([^"\']+)["\']', h) + re.findall(r'data-src=["\']([^"\']+)["\']', h)
    print("  懒加载图:", len(lo), lo[:3])
