#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.sdsx.gov.cn/channel_t_273_15736/"

result = subprocess.run([
    "curl", "-sL", "--max-time", "15",
    "-H", "User-Agent: Mozilla/5.0",
    url
], capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

title = soup.title
print("TITLE:", title.string.strip() if title else "N/A")

# Check for keywords/description
for m in [soup.find("meta", attrs={"name":"keywords"}), soup.find("meta", attrs={"name":"Keywords"})]:
    if m:
        print("KEYWORDS:", m.get("content","")[:80])

# Find all links with meaningful text
print("\n=== Links ===")
for a in soup.find_all("a", href=True):
    text = a.get_text(strip=True)
    href = a["href"]
    if len(text) > 4 and "javascript" not in href:
        span = a.find_parent("li") or a.find_parent()
        date_span = span.find("span") if span else None
        date = date_span.get_text(strip=True) if date_span else ""
        if not date:
            # Check parent for date
            for sp in (a.parent or a).find_all("span"):
                t = sp.get_text(strip=True)
                if re.search(r"\d{4}", t):
                    date = t
                    break
        print(f"  {href[:55]} | {text[:45]} | {date[:15]}")

# Paging
print("\n=== Paging ===")
for a in soup.find_all("a", href=re.compile(r"index_\d+\.html|page=\d+")):
    print(f"  {a.get_text(strip=True)} -> {a['href'][:60]}")

# Find list structure
for ul in soup.find_all("ul", class_=lambda x: x and any(c in x.lower() for c in ["list", "news", "clearfix"])):
    items = ul.find_all("li")
    if len(items) >= 3:
        print(f"\nUL class={ul.get('class','')}: {len(items)} items")
        for li in items[:3]:
            a = li.find("a")
            text = a.get_text(strip=True) if a else li.get_text(strip=True)
            print(f"  {text[:50]}")

# Check detail page
print("\n=== Detail example ===")
detail_url = soup.find("a", href=re.compile(r"doc_|detail|content"))
if detail_url:
    href = detail_url["href"]
    if href.startswith("/"):
        href = "http://www.sdsx.gov.cn" + href
    elif not href.startswith("http"):
        href = "http://www.sdsx.gov.cn/" + href.lstrip("/")
    
    result2 = subprocess.run([
        "curl", "-sL", "--max-time", "15",
        "-H", "User-Agent: Mozilla/5.0",
        href
    ], capture_output=True, text=True, timeout=30)
    html2 = result2.stdout
    soup2 = BeautifulSoup(html2, "html.parser")
    print(f"Detail URL: {href}")
    print(f"TITLE: {soup2.title.string.strip()[:80] if soup2.title else 'N/A'}")
    
    # Date
    for pat in [r"发布时间.*?(\d{4}-\d{2}-\d{2})", r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", r"(\d{4}-\d{2}-\d{2})"]:
        m = re.search(pat, html2)
        if m:
            print(f"DATE: {m.group(1)}")
            break
    
    # Content area
    for cls in ["content", "article", "text", "detail", "zoom", "editor", "main", "TRS_Editor"]:
        div = soup2.find("div", class_=lambda x: x and cls in str(x).lower())
        if div:
            txt = div.get_text(strip=True)
            if len(txt) > 50:
                print(f"Content: class={div.get('class','')} len={len(txt)}: {txt[:80]}...")
                break
    else:
        body = soup2.find("body")
        if body:
            print(f"Body text: {body.get_text(strip=True)[:100]}...")
