#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "https://www.shucheng.gov.cn/site/tpl/5186?pid=7367495&id=7367498&organId=6596321"
result = subprocess.run(
    ["curl", "-sL", "--max-time", "15",
     "-H", "User-Agent: Mozilla/5.0",
     url],
    capture_output=True, text=True, timeout=30)

html = result.stdout
soup = BeautifulSoup(html, "html.parser")
print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Scan ALL links and find the ones that look like article/content links
url_patterns = {}
for a in soup.find_all("a", href=True):
    href = a["href"]
    text = a.get_text(strip=True)
    # Identify patterns
    pattern = href.split("/")[-1].split("?")[0] if href else ""
    if len(href) > 30:
        key = "/".join(href.split("/")[:4])
        if key not in url_patterns:
            url_patterns[key] = []
        if len(url_patterns[key]) < 3:
            url_patterns[key].append((href, text[:40]))

print("\n=== URL Patterns ===")
for k, v in sorted(url_patterns.items())[:20]:
    for href, text in v:
        if text and len(text) > 3:
            print(f"  {href[:65]} | {text}")

# Find ALL li elements and their links
print("\n=== LI elements ===")
for li in soup.find_all("li"):
    a = li.find("a")
    if not a:
        continue
    href = a.get("href", "")
    text = a.get_text(strip=True)
    if len(text) < 5:
        continue
    span = li.find("span")
    date = span.get_text(strip=True) if span else ""
    print(f"  {href[:55]} | {text[:45]} | {date[:15]}")

# Check for specific content in page
print("\n=== Page structure ===")
for div in soup.find_all("div", class_=lambda x: x and any(c in x.lower() for c in ["list", "news", "info", "item", "right", "main", "content"])):
    links = div.find_all("a")
    if len(links) >= 3:
        first = links[0]
        print(f"Div class='{div.get('class','')}' - {len(links)} links - first: {first.get('href','')[:40]} | {first.get_text(strip=True)[:30]}")
