#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""verify_para_fix.py —— 验证分段修复：段落数 应等于 <p> 数"""
import importlib.util
import re

CASES = [
    ("颍泉", "/root/gov_crawler/crawl_yingquan_gggs.py",
     "https://www.yingquan.gov.cn/Content/show/1343340.html", "#zoom"),
    ("颍泉2", "/root/gov_crawler/crawl_yingquan_gggs.py",
     "https://www.yingquan.gov.cn/Content/show/1343448.html", "#zoom"),
    ("丹寨", "/root/gov_crawler/crawl_danzhai_hjzf.py",
     "https://www.qdndz.gov.cn/zwgk/zdlygk/sthj/hjzfjg/202608/t20260803_90687190.html", "#Zoom"),
    ("江城", "/root/gov_crawler/crawl_jcx_tzgg.py",
     "https://www.jcx.gov.cn/info/15974/526081.htm", "div.article_detail"),
]

for tag, script, url, sel in CASES:
    spec = importlib.util.spec_from_file_location("m_" + tag, script)
    m = importlib.util.module_from_spec(spec)
    spec.loader.exec_module(m)
    h = m.fetch(url)
    if not h:
        print("%-6s ❌ fetch 失败" % tag)
        continue
    _r = m.parse_detail(h, url)
    t, d, body = _r[0], _r[1], _r[2]   # 各站返回个数不同(3或4)，只取前三个
    content, has_table, atts = m.html_to_text(body, url)
    segs = content.split("\n\n")
    n_p = len(re.findall(r"<p[ >]", content, re.I))
    close_p = len(re.findall(r"</p\s*>", content, re.I))
    bare = [s for s in segs if not re.match(r"\s*<(?:p|table|ul|ol|h[1-6])\b", s, re.I)]
    print("=" * 92)
    print("%-6s %s" % (tag, t[:48]))
    print("  段落数=%-3d <p>=%-3d </p>=%-3d | 未被标签包裹的裸段=%d %s" % (
        len(segs), n_p, close_p, len(bare), "✅" if not bare and n_p == len(segs) else "⚠️"))
    if bare:
        for b in bare[:3]:
            print("     裸段: %s" % b[:90])
    for s in segs[:5]:
        print("     |%s" % s[:96])
