#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_yingquan2.py —— 颍泉 详情页 + 分页页数 + 列表容器定位"""
import re
import urllib.parse

import requests

requests.packages.urllib3.disable_warnings()
LIB = "https://www.yingquan.gov.cn"
URL = LIB + "/Content/showList/508/page_1.html"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"}


def get(u):
    r = requests.get(u, headers=H, timeout=30, verify=False)
    r.encoding = r.apparent_encoding or "utf-8"
    return r.status_code, r.text


print("=== ① 分页参数 pagecount ===")
_, h = get(URL)
for pat in [r'pagecount\s*=\s*"?(\d+)', r"pagecount[^>]*>(\d+)<", r'pageCount["\']?\s*[:=]\s*"?(\d+)']:
    m = re.findall(pat, h, re.I)
    if m:
        print("  %s → %s" % (pat[:36], m[:4]))
i = h.find("pagecount")
print("  原文:", re.sub(r"\s+", " ", h[max(0, i - 200):i + 300])[:500] if i >= 0 else "未找到")

print("\n=== ② 列表容器定位 ===")
i = h.find("/Content/show/")
seg = h[max(0, i - 1500):i + 200]
print("  首个条目链接前的容器片段:")
print("  ", re.sub(r"\s+", " ", seg)[-900:])

print("\n=== ③ 末页探测 ===")
for p in [1, 2, 3, 100, 999, 1000]:
    try:
        c, hh = get(LIB + "/Content/showList/508/page_%d.html" % p)
        n = len(re.findall(r'<span>\s*20\d{2}-\d{2}-\d{2}', hh))
        first = re.search(r'<span>\s*(20\d{2}-\d{2}-\d{2})\s*</span>\s*<a[^>]*title="([^"]{4,50})', hh)
        print("  page_%-4d HTTP %s 条目=%-3d 首条=%s" % (p, c, n, (first.group(1) + " " + first.group(2)[:26]) if first else "-"))
    except Exception as e:
        print("  page_%-4d ❌ %s" % (p, str(e)[:40]))

print("\n=== ④ 详情页结构 ===")
_, h1 = get(URL)
m = re.search(r'href="(/Content/show/\d+\.html)"[^>]*title="([^"]{4,60})"', h1)
if m:
    d = LIB + m.group(1)
    print("  探测:", d)
    c, hd = get(d)
    print("  HTTP:", c, "| 字节:", len(hd))
    mt = re.search(r"<title[^>]*>(.*?)</title>", hd, re.S | re.I)
    print("  <title>:", (mt.group(1).strip()[:110] if mt else "(无)"))
    for pat in [r'<meta[^>]*name=["\']?(ArticleTitle|PubDate|ColumnName|ContentSource|Author)["\']?[^>]*>']:
        mm = re.findall(pat, hd, re.I)
        if mm:
            print("  meta:", mm[:8])
    print("  各容器 <p> 数 / 文本长度:")
    blocks = re.findall(r'<div[^>]*class=["\']([^"\']+)["\'][^>]*>(.*?)</div>\s*(?=<div|</body)', hd, re.S | re.I)
    stats = [(b.count("<p"), len(re.sub(r"<[^>]+>", "", b)), cl[:38]) for cl, b in blocks]
    for pc, tx, cl in sorted(stats, reverse=True)[:10]:
        print("    p=%-3d 文本=%-6d %s" % (pc, tx, cl))
    print("  id=zoom 存在:", 'id="zoom"' in hd or "id=zoom" in hd)
    print("  class 含 cont/zoom/content 的:", re.findall(r'class="([^"]*(?:zoom|cont-con|article|content)[^"]*)"', hd, re.I)[:8])
    print("  附件:", re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"', hd, re.I)[:5])
