#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.taiqian.gov.cn/content/2026/1290326.html"
result = subprocess.run([
    "curl", "-sL", "--max-time", "15",
    "-H", "User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "-H", "Accept: text/html,application/xhtml+xml",
    url
], capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

print("TITLE:", soup.title.string.strip()[:80] if soup.title else "N/A")

# Date
for pat in [r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", r"(\d{4}-\d{2}-\d{2})"]:
    m = re.search(pat, html)
    if m: print(f"Date: {m.group(1)}"); break

# Check all div classes for content
for div in soup.find_all("div", class_=lambda x: x and len(x) > 2):
    cls = " ".join(div.get("class", []))
    txt = div.get_text(strip=True)
    links = len(div.find_all("a"))
    if len(txt) > 200 and len(txt) < 50000:
        print(f"Div class='{cls}': len={len(txt)} links={links}")
        print(f"  Preview: {txt[:80]}...")

# Attachments
for a in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", re.I)):
    print(f"Attachment: {a['href'][:50]} | {a.get_text(strip=True)[:30]}")
