#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""debug_jcx.py —— 追 <p><p> 从哪来：原始 HTML → 各阶段 repr"""
import importlib.util
import re

spec = importlib.util.spec_from_file_location("m", "/root/gov_crawler/crawl_jcx_tzgg.py")
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)

D = "https://www.jcx.gov.cn/info/15974/526231.htm"
h = m.fetch(D)
t, d, body = m.parse_detail(h, D)
print("=== 原始正文 HTML 中附件区 repr ===")
i = body.find("virtual_attach_file")
print(repr(body[max(0, i - 500):i + 260]))
print()
print("=== 原始正文里 <li>/<ul>/<p> 计数 ===")
print("  <li:", body.count("<li"), "| <ul:", body.count("<ul"), "| <p:", body.count("<p"),
      "| <p(带属性):", len(re.findall(r"<p\b[^>]+>", body)))
print()
print("=== 跑 html_to_text ===")
c, ht, atts = m.html_to_text(body, D)
print("  附件数:", len(atts))
print()
for k, seg in enumerate([p for p in c.split("\n\n") if "virtual_attach_file" in p]):
    print("  附件段%d repr:" % k)
    print("   ", repr(seg[:150]))
print()
print("  全部段落标签形状:")
for seg in c.split("\n\n"):
    head = seg[:28].replace("\n", "")
    print("    |%s|" % head)
