#!/usr/bin/env python3
import sys, re
from bs4 import BeautifulSoup
import urllib.request, ssl

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

url = "https://www.shucheng.gov.cn/site/tpl/5186?pid=7367495&id=7367498&organId=6596321"
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
html = resp.read().decode("utf-8", errors="replace")

soup = BeautifulSoup(html, "html.parser")
print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Find all links with content/detail pattern
for a in soup.find_all("a", href=re.compile(r"content/detail")):
    href = a.get("href", "")
    text = a.get_text(strip=True)
    if len(text) > 3:
        # Find parent for date
        parent = a.find_parent("li") or a.find_parent("div")
        span = parent.find("span") if parent else None
        date = span.get_text(strip=True) if span else ""
        print(f"{href[:60]} | {text[:50]} | {date}")

print("\n--- Paging ---")
for a in soup.find_all("a", href=re.compile(r"pageNum|page=")):
    print(f"{a.get('href','')} | {a.get_text(strip=True)}")

# Check detail page
print("\n--- Detail example ---")
detail_url = "https://www.shucheng.gov.cn/content/detail/66f8e1e6e42f8e1e2a8b4567.html"
req2 = urllib.request.Request(detail_url, headers={"User-Agent": "Mozilla/5.0"})
resp2 = urllib.request.urlopen(req2, timeout=30, context=ssl_ctx)
html2 = resp2.read().decode("utf-8", errors="replace")
soup2 = BeautifulSoup(html2, "html.parser")
print("Detail TITLE:", soup2.title.string.strip()[:80] if soup2.title else "N/A")

# Find content area
for cls_name in ["content", "article", "main", "text", "detail", "TRS_Editor", "ucapcontent"]:
    div = soup2.find("div", class_=lambda x: x and cls_name.lower() in x.lower()) if False else None
# Simple approach - look for divs with lots of text
divs = soup2.find_all("div", class_=lambda x: x and any(c in x.lower() for c in ["content", "article", "text", "detail", "zoom", "editor"]))
for d in divs[:5]:
    txt = d.get_text(strip=True)
    if len(txt) > 100:
        cls = d.get("class", "")
        print(f"Content: class={cls} len={len(txt)}")
        break
else:
    # Fallback
    body = soup2.find("body")
    if body:
        txt = body.get_text(strip=True)
        print(f"Body text: {txt[:200]}...")

# Date
for pat in [r"发布时间.*?(\d{4}-\d{2}-\d{2})", r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", r"(\d{4}-\d{2}-\d{2})"]:
    m = re.search(pat, html2)
    if m:
        print(f"Date: {m.group(1)}")
        break
