#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "https://www.shucheng.gov.cn/site/tpl/5186?pid=7367495&id=7367498&organId=6596321"
headers = [
    "User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Cookie: __jsluid_s=44466aca9d819cfbd85efd0d77815ac0",
]

cmd = ["curl", "-sL", "--max-time", "15"]
for h in headers:
    cmd.extend(["-H", h])
cmd.append(url)

result = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

# Check ind_gklist
for el in soup.select(".ind_gklist, #ind_gklist, div.ind_gklist"):
    print(f"ind_gklist: {el.name} text={len(el.get_text(strip=True))}")
    for child in el.find_all(recursive=False):
        print(f"  {child.name} class={child.get('class','')}")
        links = child.find_all("a", href=True)
        for a in links[:5]:
            print(f"    {a['href'][:60]} | {a.get_text(strip=True)[:40]}")
    break

# Check gklm_search
for el in soup.select(".gklm_search, #gklm_search"):
    print(f"\ngklm_search: {el.name} text={len(el.get_text(strip=True))}")
    for a in el.find_all("a", href=True)[:10]:
        print(f"  {a['href'][:60]} | {a.get_text(strip=True)[:40]}")

# Look at the whole structure
print("\n=== Full structure ===")
body = soup.find("body")
if body:
    for child in body.find_all(recursive=False):
        cls = child.get("class","")
        txt = child.get_text(strip=True)[:60]
        print(f"  {child.name}.{cls}: {txt}")
