#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""check_jcx_list.py —— 列表对账 + 下页链接链"""
import importlib.util
import re

spec = importlib.util.spec_from_file_location("m", "/root/gov_crawler/crawl_jcx_tzgg.py")
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)

h = m.fetch(m.LIST_URL)
print("页面字节:", len(h))
mark = len(re.findall(r'<li id="line_u9_\d+"', h))
dateb = len(re.findall(r'<b class="date">', h))
its = m.parse_list(h)
print("line_u9_N 标记:", mark, "| <b class=date>:", dateb, "| parse_list:", len(its))
print("对账:", "✅ 一致" if len(its) == mark == dateb else "❌")
print("标题长度 min/max:", min(len(t) for _, t, _ in its), "/", max(len(t) for _, t, _ in its))
print("含省略号标题:", [t for _, t, _ in its if "…" in t or t.endswith("...")])
print()
print("前 4 条:")
for u, t, d in its[:4]:
    print("  ", d, "|", t[:56])
print()
print("=== 下页链（4 跳）===")
url = m.LIST_URL
for i in range(4):
    hh = m.fetch(url)
    nx = m.next_page_url(hh, url)
    print("  第%d页 %-28s → 下页 %s" % (i + 1, url.replace(m.BASE_URL, ""), (nx or "(无)").replace(m.BASE_URL, "")))
    if not nx:
        break
    url = nx
