#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "https://www.shucheng.gov.cn/site/tpl/5186?pid=7367495&id=7367498&organId=6596321"
headers = [
    "User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept: text/html,application/xhtml+xml",
    "Accept-Language: zh-CN,zh;q=0.9",
    "Referer: https://www.shucheng.gov.cn/",
    "Cookie: __jsluid_s=44466aca9d819cfbd85efd0d77815ac0",
]

cmd = ["curl", "-sL", "--max-time", "15"]
for h in headers:
    cmd.extend(["-H", h])
cmd.append(url)

result = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
html = result.stdout

# Write to file for inspection
with open("/tmp/shucheng.html", "w", encoding="utf-8") as f:
    f.write(html)

soup = BeautifulSoup(html, "html.parser")
print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Look for all major content areas
print("\n=== Content areas ===")
for c in soup.find_all("div", class_=lambda x: x and len(x) > 3 and len(x) < 40):
    txt = c.get_text(strip=True)
    links = c.find_all("a", href=True)
    if len(links) >= 3 and len(txt) > 200:
        cls = c.get("class", "")
        print(f"  class={cls} links={len(links)} text={len(txt)}")
        for a in links[:3]:
            print(f"    {a['href'][:60]} | {a.get_text(strip=True)[:40]}")

# Find the list structure
print("\n=== List items ===")
for ul in soup.find_all("ul", class_=lambda x: x and any(c in x.lower() for c in ["list", "news", "info"])):
    print(f"UL class='{ul.get('class','')}' - {len(ul.find_all('li'))} items")
    for li in ul.find_all("li")[:5]:
        a = li.find("a", href=True)
        if a:
            text = a.get_text(strip=True)
            href = a["href"]
            if href.startswith("/"):
                href = "https://www.shucheng.gov.cn" + href
            span = li.find("span") or li.find("em")
            date = span.get_text(strip=True) if span else ""
            print(f"  {href[:55]} | {text[:40]} | {date}")

# Also check tables
for tbl in soup.find_all("table"):
    rows = tbl.find_all("tr")
    if len(rows) >= 3:
        print(f"\nTable: {len(rows)} rows")
        for tr in rows[:3]:
            cells = tr.find_all(["td", "th"])
            texts = [c.get_text(strip=True)[:20] for c in cells]
            print(f"  {' | '.join(texts)}")

# Paging
print("\n=== Paging ===")
for a in soup.find_all("a", href=re.compile(r"pageNum|page=|index")):
    print(f"  {a.get_text(strip=True)} -> {a['href'][:60]}")
