#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.tuoxian.gov.cn/zfxxgkzl/zfbmxxgk/gwh/gyyqgwh/zfxxgk/fdzdgknr_22198/?gk=3"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

print("TITLE:", soup.title.string.strip() if soup.title else "N/A")

# Meta keywords
for meta in soup.find_all("meta", attrs={"name": lambda x: x and "keyword" in x.lower()}):
    print("KEYWORDS:", meta.get("content","")[:80])

# Find all links with content/article patterns
print("\n=== Candidate list links ===")
for a in soup.find_all("a", href=True):
    href = a["href"]
    text = a.get_text(strip=True)
    if len(text) > 5 and "javascript" not in href:
        # Look for date nearby
        parent = a.find_parent("li") or a.find_parent("tr") or a.find_parent("div")
        date = ""
        if parent:
            for sp in parent.find_all(["span", "td", "em", "font"]):
                t = sp.get_text(strip=True)
                if re.search(r"\d{4}[-/.]\d{1,2}[-/.]\d{1,2}", t):
                    date = t
                    break
        
        # Filter for article-like patterns
        if re.search(r"content|detail|show|article|info|\d{4,}", href) and not re.search(r"css|js|png|jpg", href):
            if len(href) > 30:
                print(f"  {href[:55]} | {text[:40]} | {date[:15]}")

# Check for iframe (the URL has #iframe)
for iframe in soup.find_all("iframe"):
    print(f"\nIframe: src={iframe.get('src','')[:80]}")

# Paging
for a in soup.find_all("a", href=re.compile(r"page=|pageNum=|index_")):
    print(f"Page: {a.get_text(strip=True)} -> {a['href'][:60]}")
