#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_jy_detail.py —— 缙云详情页正文容器定位"""
import re
import subprocess

from bs4 import BeautifulSoup

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def get(u):
    p = subprocess.run(["curl", "-sk", "-L", "--max-time", "40", "-A", UA, u],
                       capture_output=True, timeout=60)
    return p.stdout.decode("utf-8", "ignore")


for d in ["https://www.jinyun.gov.cn/col/col1229425888/art/2026/art_5a4ccde6d11a4bc39b623c04a01146eb.html",
          "https://www.jinyun.gov.cn/col/col1229856959/art/2026/art_235c29d9ce074645b95027f58bec387b.html"]:
    h = get(d)
    print("=" * 96)
    print(d[-56:], "|", len(h), "字节")
    if not h:
        print("  ❌ 空")
        continue
    print("  <p> 总数:", h.count("<p"), "| <table>:", h.count("<table"), "| <img>:", h.count("<img"))
    soup = BeautifulSoup(h, "html.parser")
    print("  所有 div 的 (class/id, p数, 文本长度) 前 14:")
    stats = []
    for dv in soup.find_all("div"):
        inner = "".join(str(c) for c in dv.contents)
        tx = dv.get_text(" ", strip=True)
        cls = " ".join(dv.get("class") or []) or (dv.get("id") or "")
        stats.append((len(tx), cls, inner.count("<p"), inner))
    for tx_len, cls, pc, inner in sorted(stats, reverse=True)[:14]:
        print("     文本=%-6d p=%-3d %s" % (tx_len, pc, cls[:44]))
    best = max(stats)
    print("  最大容器 [%s] 文本: %s" % (best[1][:40], best[0]))
    # 打印最大容器的前 300 字（看是否正文）
    print("     内容:", re.sub(r"\s+", " ", BeautifulSoup(best[3], "html.parser").get_text(" ", strip=True))[:280])
    print("  附件:", re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|jpg|png))"', h, re.I)[:4])
    for k in ["ArticleTitle", "PubDate", "ColumnName"]:
        mm = re.search(r'<meta[^>]*name="%s"[^>]*content="([^"]*)"' % k, h, re.I)
        if mm:
            print("  meta %s: %s" % (k, mm.group(1)[:70]))
