#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.sdsx.gov.cn/channel_t_273_15736/"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

# Get the laypage config
for script in soup.find_all("script"):
    content = script.string or ""
    if "laypage" in content:
        print(content)
        break

# UL list items with proper date extraction
print("\n=== List items ===")
for ul in soup.find_all("ul"):
    lis = ul.find_all("li")
    items = []
    for li in lis:
        a = li.find("a", href=re.compile(r"doc_"))
        if not a:
            continue
        href = a["href"]
        title = a.get_text(strip=True)
        
        # Date is in the text of the li, not in a span
        full_text = li.get_text(" ", strip=True)
        # Remove the title from the text to get the date
        date_part = full_text.replace(title, "", 1).strip()
        
        items.append((href, title, date_part))
    
    if len(items) >= 3:
        for href, title, date in items[:10]:
            print(f"  {date[:15]:15s} | {title[:55]}")
        print(f"  ... total {len(items)} items")
        break
