#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""verify_linkbox.py —— 目标文章：详情页渲染 + 附件真实下载"""
import json
import re
import sqlite3
import urllib.request

BASE = "http://127.0.0.1:8000"
URL = "http://www.binyang.gov.cn/gk/xxgkml/shgysyjslygk/hjbhly/jsxmhjyxpjsp/t6684215.html"

db = sqlite3.connect("/root/search.db", timeout=30)
row = db.execute("SELECT id, title, content, attachments FROM gov_raw WHERE page_url=?", (URL,)).fetchone()
rid, title, content, atts = row
print("id:", rid)
print("标题:", title[:56])
print("内容里 .docx 出现:", content.count("P020260720396340403399"))
print("<table>:", content.count("<table"), "| <p>:", len(re.findall(r"<p[ >]", content, re.I)))
print("attachments:", atts[:200])

opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor())
try:
    opener.open(urllib.request.Request(BASE + "/login", data=b"username=admin",
                headers={"Content-Type": "application/x-www-form-urlencoded"}), timeout=20)
    h = opener.open(BASE + "/crawler/project/%s" % rid, timeout=30).read().decode("utf-8", "ignore")
    print("\n详情页 HTTP200 %d 字节" % len(h))
    print("  页面含附件名:", "南宁浮法玻璃" in h and "报告表" in h)
    print("  页面含 .docx 链接:", "P020260720396340403399.docx" in h)
    m = re.findall(r'href="([^"]*P020260720396340403399[^"]*)"[^>]*>([^<]{0,60})', h)
    for u, t in m[:2]:
        print("    ↳", t[:50], "|", u[:96])
except Exception as e:
    print("  ❌ 渲染:", str(e)[:70])

print("\n附件真实下载：")
att = json.loads(atts)[0]
req = urllib.request.Request(att["url"], headers={
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Referer": "http://www.binyang.gov.cn/", "Accept": "*/*"})
try:
    r = urllib.request.urlopen(req, timeout=60)
    d = r.read()
    kind = ("PDF" if d[:4] == b"%PDF" else "ZIP/OOXML(docx)" if d[:2] == b"PK" else "其他")
    print("  HTTP %s | %d 字节 | 魔数=%s | 类型=%s" % (r.getcode(), len(d), d[:4], kind))
    print("  Content-Type:", r.headers.get("Content-Type"))
    print("  文件名标注:", att["title"][-40:])
except Exception as e:
    print("  ❌", str(e)[:80])
db.close()
