#!/usr/bin/env python3
import sys, re, subprocess, json
from bs4 import BeautifulSoup

url = "https://www.shucheng.gov.cn/site/tpl/5186?pid=7367495&id=7367498&organId=6596321"

result = subprocess.run(
    ["curl", "-sL", "--max-time", "15",
     "-H", "User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
     url],
    capture_output=True, text=True, timeout=30)

html = result.stdout
soup = BeautifulSoup(html, "html.parser")
print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Find content/detail links
for a in soup.find_all("a", href=re.compile(r"content/detail")):
    href = a.get("href", "")
    text = a.get_text(strip=True)
    if len(text) < 4:
        continue
    # Complete URL
    if href.startswith("/"):
        href = "https://www.shucheng.gov.cn" + href
    elif not href.startswith("http"):
        href = "https://www.shucheng.gov.cn/" + href.partition("/")[2] if "/" in href else href
    
    # Find date from nearby span
    date = ""
    parent = a.find_parent("li") or a.find_parent("div", class_=lambda x: x and "list" in x.lower())
    if parent:
        for sp in parent.find_all("span"):
            t = sp.get_text(strip=True)
            if re.search(r"\d{4}", t):
                date = t
                break
    print(f"{href[:60]} | {text[:50]} | {date}")

# Check paging
print("\n--- Paging ---")
for a in soup.find_all("a", href=re.compile(r"pageNum|page=")):
    href = a.get("href", "")
    if href.startswith("/"):
        href = "https://www.shucheng.gov.cn" + href
    print(f"  {a.get_text(strip=True)} -> {href[:60]}")

# Check total items on page
all_links = soup.find_all("a", href=re.compile(r"content/detail"))
print(f"\nTotal content links: {len(all_links)}")

# Check a detail page
print("\n--- Detail page ---")
detail_url = "https://www.shucheng.gov.cn/content/detail/66f8e1e6e42f8e1e2a8b4567.html"
result2 = subprocess.run(
    ["curl", "-sL", "--max-time", "15",
     "-H", "User-Agent: Mozilla/5.0",
     detail_url],
    capture_output=True, text=True, timeout=30)
html2 = result2.stdout
soup2 = BeautifulSoup(html2, "html.parser")
print("TITLE:", soup2.title.string.strip()[:80] if soup2.title else "N/A")

# Find content
for cls in ["content", "article", "text", "detail", "zoom", "editor", "main", "body"]:
    divs = soup2.find_all("div", class_=lambda x: x and cls in x.lower()) if cls else []
    for d in divs:
        txt = d.get_text(strip=True)
        if len(txt) > 100:
            print(f"Content div class='{d.get('class','')}' len={len(txt)}: {txt[:80]}...")
            break
    else:
        continue
    break

# Date
for pat in [r"发布时间.*?(\d{4}-\d{2}-\d{2})", r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", r"(\d{4}-\d{2}-\d{2})"]:
    m = re.search(pat, html2)
    if m:
        print(f"Date: {m.group(1)}")
        break

# Attachments
for a in soup2.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", re.I)):
    print(f"Attachment: {a['href'][:60]} | {a.get_text(strip=True)[:30]}")
