#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""diag_fufa.py —— 南宁浮法玻璃那条：库内正文尾部 vs 源站原文的表格/附件区"""
import json
import re
import sqlite3
import subprocess

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def get(u):
    p = subprocess.run(["curl", "-sk", "-L", "--max-time", "40", "-A", UA, u],
                       capture_output=True, timeout=60)
    return p.stdout.decode("utf-8", "ignore")


db = sqlite3.connect("/root/search.db", timeout=30)
db.execute("PRAGMA busy_timeout=30000")
rows = list(db.execute(
    "SELECT id, site_name, script_name, page_url, title, content, attachments, has_table "
    "FROM gov_raw WHERE title LIKE '%浮法玻璃%' OR title LIKE '%危险废物暂存间%'"))
print("命中 %d 条" % len(rows))
for rid, site, script, url, title, content, atts, ht in rows:
    print("=" * 100)
    print("id=%s | %s | script=%s" % (rid, site, script))
    print("标题:", title)
    print("URL:", url)
    print("has_table=%s | attachments=%s" % (ht, (atts or "(空)")[:160]))
    body = re.sub(r"<[^>]+>", "", content or "")
    print("库内正文: %d 字符(纯文本) | <p>=%d | <table>=%d | <a>=%d" % (
        len(body), len(re.findall(r"<p[ >]", content or "", re.I)),
        len(re.findall(r"<table", content or "", re.I)), len(re.findall(r"<a ", content or "", re.I))))
    print("  尾部 300 字:", repr((content or "")[-300:]))
    if not url:
        continue
    h = get(url)
    print("\n  --- 源站 %d 字节 ---" % len(h))
    # 找正文容器
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(h, "html.parser")
    for sel in ["div.trs_editor_view", "div.ue_table", "div.content", "div#zoom"]:
        el = soup.select_one(sel)
        if el is None:
            print("   %-24s 不存在" % sel)
            continue
        inn = "".join(str(c) for c in el.contents)
        print("   %-24s p=%-3d table=%-2d a=%-3d 文本=%-5d" % (
            sel, inn.count("<p"), inn.count("<table"), inn.count("<a "), len(el.get_text(" ", strip=True))))
    # 全页表格与附件
    tabs = soup.find_all("table")
    print("   全页 <table> 数:", len(tabs))
    for t in tabs:
        tx = re.sub(r"\s+", " ", t.get_text(" ", strip=True))
        print("     表格(%d字): %s" % (len(tx), tx[:150]))
        for a in t.find_all("a", href=True):
            print("        ↳ 表内链接:", (a.get("title") or a.get_text(" ", strip=True))[:40], "|", a["href"][:80])
    atts_src = re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd|jpg|png))"', h, re.I)
    print("   全页按扩展名附件:", len(atts_src), atts_src[:4])
    # 附件区容器名候选
    for dv in soup.find_all("div"):
        cls = " ".join(dv.get("class") or [])
        if re.search(r"attach|file|download|fj|fujian|ue_table", cls, re.I):
            tx = re.sub(r"\s+", " ", dv.get_text(" ", strip=True))
            print("   ⭐附件候选div [%s] (%d字): %s" % (cls[:40], len(tx), tx[:120]))
db.close()
