#!/usr/bin/env python3
"""Debug qx detail page - self contained."""
import requests, re, sys, os
sys.path.insert(0, '/root/gov_crawler')

url = "https://www.qx.gov.cn/xxgk2222222222222222222222222222222222222/20260720/30310684.html"
s = requests.Session()
s.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
})
resp = s.get(url, timeout=30)
resp.encoding = 'utf-8'
html = resp.text
print(f"Status: {resp.status_code}, Length: {len(html)}")

m = re.search(r"<title>(.*?)</title>", html)
if m: print(f"Title tag: {m.group(1)}")

m = re.search(r"<h1[^>]*>(.*?)</h1>", html, re.DOTALL)
if m: print(f"H1: {m.group(1)[:100]}")

idx = html.find('id="Zoom"')
if idx != -1:
    print(f"Zoom found at {idx}")
    print(html[idx:idx+500])
else:
    print("No Zoom div")
    # Check if page has a different structure
    for cls in ["detail", "content", "article", "main"]:
        i = html.find(f'class="{cls}')
        if i != -1:
            print(f"Found class '{cls}' at {i}: {html[i:i+200]}")
            break

# Check the first detail that was inserted
print("\n--- Checking DB ---")
os.system("sqlite3 /mnt/data/search.db \"SELECT title FROM gov_raw WHERE page_url LIKE '%qx.gov.cn/xxgk222%' LIMIT 3\"")
