#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

# Check a real detail page
url = "http://www.taiqian.gov.cn/content/2026/1290326.htm"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

print("TITLE:", soup.title.string.strip()[:80] if soup.title else "N/A")

# Date
for pat in [r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", r"(\d{4}-\d{2}-\d{2})"]:
    m = re.search(pat, html)
    if m: print(f"Date: {m.group(1)}"); break

# Content area
for cls_name in ["content", "article", "text", "detail", "zoom", "editor", "TRS_Editor", "main_top"]:
    div = soup.find("div", class_=lambda x: x and cls_name in str(x).lower())
    if div:
        txt_len = len(div.get_text(strip=True))
        if txt_len > 50:
            print(f"Content div: class={div.get('class','')} len={txt_len}")
            print(f"  Preview: {div.get_text(strip=True)[:80]}...")
            break

# Also check the main_top_content
div = soup.find("div", class_="main_top_content")
if div:
    print(f"main_top_content len={len(div.get_text(strip=True))}")

# Check page total info
print("\n=== Total info ===")
r2 = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0",
    "http://www.taiqian.gov.cn/channel/list/19256.html"],
    capture_output=True, text=True, timeout=15)
m = re.search(r"共(\d+)条", r2.stdout)
if m: print(f"Total: {m.group(1)}条")
m = re.search(r"(\d+)页", r2.stdout)
for m in re.finditer(r"href=[\"']([^\"']*pn=(\d+))[\"']", r2.stdout):
    print(f"Last page: pn={m.group(2)}")
    break
