#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_jcx3.py —— 附件区 + 跟随下页链接可行性 + 多篇正文形态"""
import re
import urllib.parse

import requests

requests.packages.urllib3.disable_warnings()
LIB = "https://www.jcx.gov.cn"
URL = LIB + "/xwzx/tzgg.htm"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"}


def get(u):
    r = requests.get(u, headers=H, timeout=30, verify=False)
    r.encoding = r.apparent_encoding or "utf-8"
    return r.status_code, r.text


print("=== ① 「下页」链接跟随（page1 → 逐页）===")
u = URL
seen = []
for step in range(4):
    c, h = get(u)
    n = len(re.findall(r'<li id="line_u9_\d+"', h))
    d = re.search(r'<b class="date">(20\d\d-\d\d-\d\d)</b>', h)
    nav = re.search(r'共(\d+)条\s+(\d+)/(\d+)', h)
    nxt = re.findall(r'<a href="([^"]+)" class="Next"[^>]*>下页</a>', h)
    print("  步%d %-42s HTTP %s li=%s 首=%s 页码=%s 下页=%s" % (
        step, u.replace(LIB, ""), c, n, d.group(1) if d else "-",
        (nav.group(2) + "/" + nav.group(3)) if nav else "-", nxt[0] if nxt else "(无)"))
    seen.append(u)
    if not nxt:
        print("   → 无「下页」链接，停止")
        break
    u = urllib.parse.urljoin(u, nxt[0])

print("\n=== ② 多篇详情：正文容器 + 附件区 ===")
_, h0 = get(URL)
links = re.findall(r'<a href="(\.\./info/\d+/\d+\.htm)"[^>]*title="([^"]*)"', h0)[:5]
for href, t in links:
    d = urllib.parse.urljoin(URL, href)
    c, hd = get(d)
    m = re.search(r'<div[^>]*class="article_detail"[^>]*>(.*?)</div>\s*<div', hd, re.S)
    body = m.group(1) if m else ""
    tx = re.sub(r"<[^>]+>", "", body)
    files = re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd))"', hd, re.I)
    dl = re.findall(r'href="([^"]*(?:download|attach|/upload|/file)[^"]*)"', hd, re.I)
    print("\n  ── %s" % t[:46])
    print("     HTTP %s 字节=%d | 正文 p=%d 文本=%d | 正文内<a>=%d <table>=%d <img>=%d" % (
        c, len(hd), body.count("<p"), len(tx), body.count("<a "), body.lower().count("<table"),
        body.lower().count("<img")))
    print("     全页文件链接:", len(files), files[:3])
    print("     download/attach 类链接:", len(dl), dl[:3])
    # 找附件区关键词
    for kw in ["附件", "下载", "downloadfile", "fujian", "附件下载"]:
        i = hd.find(kw)
        if i > 0:
            print("     含「%s」@%d: %s" % (kw, i, re.sub(r"\s+", " ", hd[max(0, i - 120):i + 220])[:300]))
            break
    if body:
        print("     正文起头:", re.sub(r"\s+", " ", tx)[:110])
