#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.sdsx.gov.cn/channel_t_273_15736/"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Find list structure - look at ULs
for ul in soup.find_all("ul"):
    lis = ul.find_all("li")
    items = []
    for li in lis:
        a = li.find("a", href=re.compile(r"doc_"))
        if not a:
            continue
        href = a["href"]
        title = a.get_text(strip=True)
        # Find date
        span = li.find("span") or li.find("em") or li.find("time")
        date = span.get_text(strip=True) if span else ""
        items.append((href, title, date))
    
    if len(items) >= 3:
        print(f"\nUL ({len(items)} items):")
        for href, title, date in items[:5]:
            print(f"  {href[:50]} | {title[:45]} | {date}")
        break

# Check for JS paging
for script in soup.find_all("script"):
    content = script.string or ""
    if "page" in content.lower() or "createPage" in content or "Page" in content:
        lines = [l.strip() for l in content.split("\n") if "page" in l.lower() or "Page" in l]
        for l in lines[:5]:
            print(f"  JS page: {l[:120]}")

# Total count
m = re.search(r"共(\d+)条", html)
if m:
    print(f"Total items: {m.group(1)}")
m = re.search(r"共(\d+)页", html)
if m:
    print(f"Total pages: {m.group(1)}")

# Page links
for a in soup.find_all("a", href=True):
    href = a["href"]
    if "index" in href and ("html" in href or "shtml" in href):
        print(f"Page link: {a.get_text(strip=True)} -> {href}")

# Info bar
for div in soup.find_all("div", class_=lambda x: x and any(c in x.lower() for c in ["info", "page", "pageing", "pagination"])):
    print(f"Info div: class={div.get('class','')} text={div.get_text(strip=True)[:100]}")
