#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.taiqian.gov.cn/channel/list/19256.html"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Meta keywords
for meta in soup.find_all("meta", attrs={"name": lambda x: x and "keyword" in x.lower()}):
    print("KEYWORDS:", meta.get("content","")[:80])

# Find the channel name
for div in soup.find_all("div", class_=lambda x: x and any(c in x.lower() for c in ["channel", "title", "location", "nav", "crumb", "path"])):
    txt = div.get_text(strip=True)
    if txt and len(txt) > 5:
        print(f"Channel: {txt[:80]}")

# Find list items
for ul in soup.find_all("ul", class_=lambda x: x and any(c in x.lower() for c in ["list", "news", "info", "article"])):
    items = []
    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a: continue
        href = a["href"]
        text = a.get_text(strip=True)
        if len(text) < 5: continue
        span = li.find("span") or li.find("em") or li.find("time")
        date = span.get_text(strip=True) if span else ""
        items.append((href, text, date))
    if len(items) >= 3:
        print(f"\nUL class={ul.get('class','')}: {len(items)} items")
        for href, text, date in items[:5]:
            print(f"  {href[:50]} | {text[:40]} | {date}")
        break

# Find ALL article-like links
print("\n=== All doc links ===")
for a in soup.find_all("a", href=re.compile(r"doc_|content|detail|show|art|html/\d+")):
    href = a["href"]
    text = a.get_text(strip=True)
    if len(text) > 5:
        parent = a.find_parent("li") or a.find_parent()
        span = parent.find("span") if parent else None
        date = span.get_text(strip=True)[:15] if span else ""
        print(f"  {href[:45]} | {text[:40]} | {date}")

# Paging
print("\n=== Paging ===")
for a in soup.find_all("a", href=re.compile(r"page=|index_|list.*\d")):
    print(f"  {a.get_text(strip=True)} -> {a['href'][:60]}")

# Check for total info
m = re.search(r"共(\d+)条", html)
if m: print(f"Total: {m.group(1)}条")
m = re.search(r"共(\d+)页", html)
if m: print(f"Pages: {m.group(1)}页")

# Check detail page
detail_a = soup.find("a", href=re.compile(r"detail|content"))
if detail_a:
    href = detail_a["href"]
    if href.startswith("/"):
        href = "http://www.taiqian.gov.cn" + href
    elif not href.startswith("http"):
        href = "http://www.taiqian.gov.cn/" + href.lstrip("/")
    
    result2 = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", href],
        capture_output=True, text=True, timeout=30)
    html2 = result2.stdout
    soup2 = BeautifulSoup(html2, "html.parser")
    print(f"\nDetail: {href}")
    print(f"  Title: {soup2.title.string.strip()[:60] if soup2.title else 'N/A'}")
    for pat in [r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", r"(\d{4}-\d{2}-\d{2})"]:
        m = re.search(pat, html2)
        if m: print(f"  Date: {m.group(1)}"); break
    for cls in ["content", "article", "text", "detail", "zoom", "editor", "TRS_Editor", "main"]:
        div = soup2.find("div", class_=lambda x: x and cls in str(x).lower())
        if div and len(div.get_text(strip=True)) > 50:
            print(f"  Content: class={div.get('class','')} len={len(div.get_text(strip=True))}")
            break
