#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""diag_fufa2.py —— 精确定位 南宁浮法玻璃 危险废物暂存间 那条"""
import re
import sqlite3
import subprocess

from bs4 import BeautifulSoup

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def get(u):
    p = subprocess.run(["curl", "-sk", "-L", "--max-time", "40", "-A", UA, u],
                       capture_output=True, timeout=60)
    return p.stdout.decode("utf-8", "ignore")


db = sqlite3.connect("/root/search.db", timeout=30)
db.execute("PRAGMA busy_timeout=30000")
rows = list(db.execute(
    "SELECT id, site_name, script_name, page_url, title, content, attachments, has_table "
    "FROM gov_raw WHERE title LIKE '%南宁浮法玻璃%'"))
print("命中 %d 条\n" % len(rows))
for rid, site, script, url, title, content, atts, ht in rows:
    print("=" * 100)
    print("id=%s | site=%s | script=%s" % (rid, site, script))
    print("标题:", title)
    print("URL:", url)
    c = content or ""
    print("has_table=%s | attachments=%s" % (ht, (atts or "(空)")[:200]))
    print("库内: 纯文本 %d 字 | <p>=%d | <table>=%d | <a>=%d" % (
        len(re.sub(r"<[^>]+>", "", c)), len(re.findall(r"<p[ >]", c, re.I)),
        len(re.findall(r"<table", c, re.I)), len(re.findall(r"<a ", c, re.I))))
    print("  结尾 400 字:", repr(c[-400:]))

    h = get(url)
    print("\n  --- 源站 %d 字节 ---" % len(h))
    soup = BeautifulSoup(h, "html.parser")
    # 正文候选
    for sel in ["div.trs_editor_view", "div.ue_table", "div.content", "div.article", "div#zoom"]:
        el = soup.select_one(sel)
        if el is None:
            print("   %-26s ❌" % sel)
            continue
        inn = "".join(str(x) for x in el.contents)
        print("   %-26s p=%-3d table=%-2d a=%-3d 文本=%-5d" % (
            sel, inn.count("<p"), inn.count("<table"), inn.count("<a "), len(el.get_text(" ", strip=True))))
    # 该文的 trs_editor_view 内部结构（直接子节点）
    el = soup.select_one("div.trs_editor_view")
    if el is not None:
        print("   trs_editor_view 直接子节点:")
        for ch in el.children:
            nm = getattr(ch, "name", None)
            if nm:
                cls = " ".join(ch.get("class") or []) if hasattr(ch, "get") else ""
                print("      <%s class=%s> %s" % (nm, cls[:34], re.sub(r"\s+", " ", ch.get_text(" ", strip=True))[:90]))
            elif str(ch).strip():
                print("      (text) %s" % re.sub(r"\s+", " ", str(ch))[:80])
    # 全页表格
    print("   全页 <table>:", len(soup.find_all("table")))
    for t in soup.find_all("table"):
        tx = re.sub(r"\s+", " ", t.get_text(" ", strip=True))
        print("     表格(%d字) 父容器=%s: %s" % (
            len(tx), " ".join((t.parent.get("class") or [])) if t.parent else "?", tx[:140]))
        for a in t.find_all("a", href=True):
            print("        ↳", (a.get("title") or a.get_text(" ", strip=True))[:44], "|", a["href"][:90])
    print("   全页扩展名附件:", re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd))"', h, re.I)[:4])
db.close()
