#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""verify_para_fix_live.py —— 验证库内已修复行 + app 渲染 + 同步面"""
import json
import re
import sqlite3
import urllib.request

BASE = "http://127.0.0.1:8000"
db = sqlite3.connect("/root/search.db", timeout=30)
db.execute("PRAGMA busy_timeout=30000")

URL = "https://www.yingquan.gov.cn/Content/show/1343340.html"
r = db.execute("SELECT id, title, content FROM gov_raw WHERE page_url=?", (URL,)).fetchone()
print("=== ① 库内该行现状 ===")
print("  id:", r[0], "| 标题:", r[1][:40])
c = r[2]
segs = [x for x in c.split("\n\n") if x.strip()]
print("  段落数:", len(segs), "| <p>:", len(re.findall(r"<p[ >]", c, re.I)),
      "|  </p>:", len(re.findall(r"</p\s*>", c, re.I)), "| 长度:", len(c))
print("  前 3 段:")
for s in segs[:3]:
    print("    ", s[:100])

print("\n=== ② app 详情页渲染 ===")
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor())
try:
    opener.open(urllib.request.Request(BASE + "/login", data=b"username=admin",
                headers={"Content-Type": "application/x-www-form-urlencoded"}), timeout=20)
    h = opener.open(BASE + "/crawler/project/%s" % r[0], timeout=30).read().decode("utf-8", "ignore")
    print("  HTTP 200 | 字节", len(h))
    # 渲染出的 <p> 数（页面里该文章的段落）
    body_ps = len(re.findall(r"<p[ >]", h, re.I))
    print("  页面 <p> 数:", body_ps)
    for key in ["一、项目简述", "项目名称：年产能7600吨", "建设单位：安徽福方高科"]:
        # 判断每个关键句是否在独立 <p> 里
        m = re.search(r"<p[^>]*>([^<]{0,200})", h)
        print("  含「%s」: %s" % (key, key in h))
    for mm in list(re.finditer(r"<p[^>]*>(.{0,60})", h))[:6]:
        print("    |", re.sub(r"\s+", " ", mm.group(1))[:70])
except Exception as e:
    print("  ❌", str(e)[:90])
db.close()
