#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""batch6_fix12.py —— 查大足短正文 + 修宾阳标题后缀"""
import importlib.util
import json
import re
import subprocess

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")


def get(u):
    p = subprocess.run(["curl", "-sk", "-L", "--max-time", "40", "-A", UA, u],
                       capture_output=True, timeout=60)
    return p.stdout.decode("utf-8", "ignore")


print("=== ① 大足：3 条短正文的实际形态 ===")
spec = importlib.util.spec_from_file_location("m", "/root/gov_crawler/crawl_dazu_glz_qtgw.py")
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)
rows = [json.loads(l) for l in open("/tmp/b6_crawl_dazu_glz_qtgw.py.jsonl", encoding="utf-8")]
short = [r for r in rows if len(re.sub(r"<[^>]+>", "", r["content"]).strip()) < 120 and not r["attachments"]]
print("  短正文条数:", len(short))
for r in short:
    print("\n  【%s】%s" % (r["pub_date"], r["title"][:44]))
    print("     URL:", r["source_url"])
    print("     正文:", repr(r["content"][:200]))
    h = get(r["source_url"])
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(h, "html.parser")
    for sel in ["div.zwxl-content div.trs_editor_view", "div.trs_editor_view", "div.zwxl-content",
                "div.pages_content", "div#zoom", "div.article"]:
        el = soup.select_one(sel)
        if el is not None:
            tx = el.get_text(" ", strip=True)
            print("     %-38s p=%d 文本=%d" % (sel, "".join(str(c) for c in el.contents).count("<p"), len(tx)))
    atts = re.findall(r'href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|jpg|png))"', h, re.I)
    print("     全页文件链接:", len(atts), atts[:3])
    print("     含 trs_editor:", "trs_editor" in h.lower(), "| 页面字节:", len(h))
