#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fix_residual_row.py —— 修剩下的 1 行 <p><p>（重抓 + UPDATE）并复验三分段判据"""
import importlib.util
import json
import re
import sqlite3

SITE = "颍泉区人民政府-公告公示"
db = sqlite3.connect("/root/search.db", timeout=60)
db.execute("PRAGMA busy_timeout=60000")

spec = importlib.util.spec_from_file_location("m", "/root/gov_crawler/crawl_yingquan_gggs.py")
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)

print("=== ① 修 <p><p> 行 ===")
rows = list(db.execute("SELECT id, page_url, title FROM gov_raw WHERE site_name=? AND content LIKE '%<p><p%'", (SITE,)))
print("待修: %d 行" % len(rows))
for rid, url, t in rows:
    h = m.fetch(url)
    if not h:
        print("  ❌", url); continue
    r = m.parse_detail(h, url)
    body, extra = r[2], (r[3] if len(r) > 3 else [])
    content, has_table, atts = m.html_to_text(body, url)
    known = set(u for u, _ in atts)
    for u, nm in extra:
        if u in known:
            continue
        known.add(u); atts.append((u, nm))
        content += '\n\n<p><a href="%s" target="_blank">%s</a></p>' % (u, nm)
    db.execute("UPDATE gov_raw SET content=?, attachments=?, has_table=? WHERE page_url=?",
               (content, json.dumps([{"title": x, "url": u} for u, x in atts], ensure_ascii=False) if atts else "",
                has_table, url))
    db.commit()
    print("  ✅ %s" % (t or "")[:40])
    print("     残留<p><p>: %s | <p>数: %d" % ("<p><p" in content, len(re.findall(r"<p[ >]", content, re.I))))

print()
print("=== ② 三站三分段判据复验 ===")
for S in ["颍泉区人民政府-公告公示", "丹寨县人民政府-环境执法监管", "江城县人民政府-通知公告"]:
    tot = db.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (S,)).fetchone()[0]
    flatten = 0
    for rid, c, t in db.execute("SELECT id, content, title FROM gov_raw WHERE site_name=?", (S,)):
        c = c or ""
        if c.lstrip().startswith("<table"):
            continue          # 纯表格正文，无 <p> 属正常
        segs = [x for x in c.split("\n\n") if x.strip()]
        n_p = len(re.findall(r"<p[ >]", c, re.I))
        if len(segs) >= 3 and n_p < len(segs) * 0.7:
            flatten += 1
            print("   ⚠️ %s | 段=%d <p>=%d | %s" % (S[:12], len(segs), n_p, (t or "")[:34]))
    nested = db.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=? AND content LIKE '%<p><p%'", (S,)).fetchone()[0]
    nop = db.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=? AND LENGTH(content)>400 "
                     "AND content NOT LIKE '%<p%' AND content NOT LIKE '%<table%'", (S,)).fetchone()[0]
    print("  【%s】%d 行 | 拍平 %d | 嵌套<p><p> %d | 长正文无<p>也无table %d" % (S, tot, flatten, nested, nop))
db.close()
