#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "https://www.shucheng.gov.cn/site/tpl/5186?pid=7367495&id=7367498&organId=6596321"

result = subprocess.run([
    "curl", "-sL", "--max-time", "15",
    "-H", "User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "-H", "Accept: text/html,application/xhtml+xml",
    "-H", "Accept-Language: zh-CN,zh;q=0.9",
    url
], capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

title = soup.title
print("TITLE:", title.string.strip() if title else "N/A (title not found)")
if title:
    print("  raw:", repr(title.string.strip()))

# Check if it's an error page
if "403" in html[:500] or "创宇盾" in html or "CloudWAF" in html:
    print("WAF BLOCKED")
    sys.exit(1)

# Find all links with titles
print("\n=== Links with text > 5 chars ===")
for a in soup.find_all("a", href=True):
    text = a.get_text(strip=True)
    href = a["href"]
    if len(text) > 5 and "javascript" not in href:
        # Check if near a span with date
        parent = a.find_parent() or a
        span = parent.find("span")
        date = span.get_text(strip=True) if span else ""
        print(f"  {href[:55]} | {text[:45]} | {date[:15]}")

# Find page URL via the .js dynamic loading
print("\n=== Script-based content ===")
for script in soup.find_all("script"):
    content = script.string or ""
    if "src=" in content and "organId" in content:
        lines = content.strip().split("\n")
        for l in lines:
            l = l.strip()
            if l:
                print(f"  {l[:120]}")
