#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.tuoxian.gov.cn/zfxxgkzl/zfbmxxgk/gwh/gyyqgwh/zfxxgk/fdzdgknr_22198/?gk=3"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

# Get the iframe
iframe = soup.find("iframe", id="iframe")
if iframe:
    src = iframe.get("src", "")
    print(f"Iframe src: {src}")
    # Try to fetch iframe content
    if src:
        if src.startswith("/"):
            src = "http://www.tuoxian.gov.cn" + src
        elif not src.startswith("http"):
            src = "http://www.tuoxian.gov.cn/" + src.lstrip("/")
        
        result2 = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", src],
            capture_output=True, text=True, timeout=30)
        html2 = result2.stdout
        soup2 = BeautifulSoup(html2, "html.parser")
        print(f"Iframe title: {soup2.title.string.strip() if soup2.title else 'N/A'}")
        
        # Find content links
        for a in soup2.find_all("a", href=True):
            href = a["href"]
            text = a.get_text(strip=True)
            if len(text) > 5 and "javascript" not in href:
                parent = a.find_parent("li") or a.find_parent("tr") or a.find_parent("div")
                date = ""
                if parent:
                    for sp in parent.find_all(["span", "td", "em"]):
                        t = sp.get_text(strip=True)
                        if re.search(r"\d{4}[-/.]", t):
                            date = t
                            break
                print(f"  {href[:50]} | {text[:40]} | {date[:15]}")
        
        # Paging
        for a in soup2.find_all("a", href=re.compile(r"page=|pageNum=")):
            print(f"Page: {a.get_text(strip=True)} -> {a['href'][:50]}")
        
        print(f"\nIframe size: {len(html2)} bytes")
else:
    print("No iframe found")
    # Check xxgkItemList
    div = soup.find("div", class_="xxgkItemList")
    if div:
        print(f"xxgkItemList: {div.get_text(strip=True)[:200]}")
