#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_jcx4.py —— BeautifulSoup 精确定位 article_detail 与附件链接归属"""
import re
import urllib.parse

import requests
from bs4 import BeautifulSoup

requests.packages.urllib3.disable_warnings()
LIB = "https://www.jcx.gov.cn"
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"}


def get(u):
    r = requests.get(u, headers=H, timeout=30, verify=False)
    r.encoding = r.apparent_encoding or "utf-8"
    return r.text


for d in ["https://www.jcx.gov.cn/info/15974/526231.htm",
          "https://www.jcx.gov.cn/info/15974/525971.htm",
          "https://www.jcx.gov.cn/info/15974/526081.htm"]:
    h = get(d)
    soup = BeautifulSoup(h, "html.parser")
    print("=" * 88)
    print("URL:", d)
    print("  article_detail 命中数:", len(soup.select("div.article_detail")))
    el = soup.select_one("div.article_detail")
    if el:
        inner = "".join(str(c) for c in el.contents)
        tx = el.get_text(" ", strip=True)
        print("  article_detail: <p>=%d 文本=%d <table>=%d <img>=%d <div>=%d" % (
            inner.count("<p"), len(tx), inner.lower().count("<table"),
            inner.lower().count("<img"), inner.lower().count("<div")))
        print("  文本起头:", tx[:100])
        print("  文本结尾:", tx[-120:])
        # 附件链接是否在容器内
        al = el.find_all("a", href=True)
        print("  容器内 <a> 数:", len(al))
        for a in al[:6]:
            print("     ", (a.get_text(" ", strip=True) or "(空)")[:50], "→", a["href"][:100])
    # 全页 vsb 附件链接
    vsb = [a["href"] for a in soup.find_all("a", href=True) if "virtual_attach_file" in a["href"]]
    print("  全页 virtual_attach_file 链接:", len(vsb))
    for v in vsb[:4]:
        print("     ", v[:110])
    # 是否有独立附件区
    for sel in ["div.fujian", "div.attachments", "#fujian", "div.xiazai", "div.download",
                "div.article_attach", "div.annex"]:
        if soup.select_one(sel):
            print("  附件区 selector 命中:", sel)
    # nextList 内容（要排除）
    nl = soup.select_one("div.nextList")
    if nl:
        print("  nextList(要排除):", nl.get_text(" ", strip=True)[:80])
    # 正文里有无 table（行政许可类常有）
    if el:
        for t in el.find_all("table")[:2]:
            print("  正文表格行数:", len(t.find_all("tr")))
