#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""diag_fufa3.py —— 看「见附」那段的原始 HTML + 逐步过 html_to_text 定位丢失点"""
import importlib.util
import re
import subprocess

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
URL = "http://www.binyang.gov.cn/gk/xxgkml/shgysyjslygk/hjbhly/jsxmhjyxpjsp/t6684215.html"

h = subprocess.run(["curl", "-sk", "-L", "--max-time", "40", "-A", UA, URL],
                   capture_output=True, timeout=60).stdout.decode("utf-8", "ignore")

print("=== ① 源站「见附」附近原始字节 ===")
for m in re.finditer("见附", h):
    i = m.start()
    print(repr(h[max(0, i - 500):i + 350]))
    print("-" * 90)

print("\n=== ② 源站 .docx 链接的原始形态 ===")
for m in re.finditer(r"P020260720396340403399\.docx", h):
    i = m.start()
    print(repr(h[max(0, i - 300):i + 120]))
    print("-" * 90)

print("\n=== ③ 跑脚本的 html_to_text 看产出 ===")
spec = importlib.util.spec_from_file_location("m", "/root/gov_crawler/crawl_binyang_hjsp.py")
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)
r = m.parse_detail(h, URL)
body = r[2]
content, has_table, atts = m.html_to_text(body, URL)
print("has_table=%s | attachments=%s" % (has_table, atts))
print("content 尾部 400:", repr(content[-400:]))
print("content 里含 .docx:", "P020260720396340403399" in content)
bb = body if isinstance(body, str) else str(body)
print("传入的 body 里含 .docx:", "P020260720396340403399" in bb)
i = bb.find("见附")
print("body 片段:", repr(bb[max(0, i - 400):i + 200]) if i > 0 else "(body 里没有「见附」)")
