#!/usr/bin/env python3
import sys, re, subprocess, json
from bs4 import BeautifulSoup

url = "http://www.tuoxian.gov.cn/zfxxgkzl/zfbmxxgk/gwh/gyyqgwh/zfxxgk/fdzdgknr_22198/?gk=3"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

# Find all unique element IDs and classes
ids = set()
classes = set()
for el in soup.find_all(True):
    if el.get("id"): ids.add(el["id"])
    if el.get("class"): classes.update(el.get("class"))
print("IDs:", sorted(ids)[:30])
print("Classes:", sorted(classes)[:40])

# Find the main content div
for el in soup.find_all(["div", "section", "main", "article"]):
    tag_id = el.get("id", "")
    tag_class = " ".join(el.get("class", []))
    text_len = len(el.get_text(strip=True))
    links = len(el.find_all("a"))
    if text_len > 200 or links > 5:
        if "nav" not in tag_class.lower() and "foot" not in tag_class.lower():
            print(f"\n{el.name} id={tag_id} class={tag_class[:50]} text={text_len} links={links}")
            # Show first few links
            for a in el.find_all("a", href=True)[:5]:
                print(f"  {a['href'][:50]} | {a.get_text(strip=True)[:30]}")
