#!/usr/bin/env python3
import sys, re, subprocess
from bs4 import BeautifulSoup

url = "http://www.tuoxian.gov.cn/zfxxgkzl/zfbmxxgk/gwh/gyyqgwh/zfxxgk/fdzdgknr_22198/?gk=3"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
html = result.stdout
soup = BeautifulSoup(html, "html.parser")

# Find ALL script content
for i, script in enumerate(soup.find_all("script")):
    content = script.string or ""
    if len(content) > 100:
        # Check for AJAX, getURL, loadURL etc
        keywords = ["url", "ajax", "load", "src", "page", "gk=", "cat=", "data", "list"]
        lines = content.split("\n")
        relevant = []
        for line in lines:
            if any(k in line.lower() for k in keywords):
                relevant.append(line.strip()[:150])
        if relevant:
            print(f"=== Script {i} ({len(content)} chars) ===")
            for l in relevant[:10]:
                print(l)

# Check the xxgkItemList links more closely
div = soup.find("div", class_="xxgkItemList")
if div:
    for a in div.find_all("a", href=True):
        print(f"Cat item: {a['href'][:50]} | {a.get_text(strip=True)}")

# Check what gk=1,2,3 etc mean
print("\n=== Try different gk params ===")
for gk in range(1, 10):
    test_url = f"http://www.tuoxian.gov.cn/zfxxgkzl/zfbmxxgk/gwh/gyyqgwh/zfxxgk/fdzdgknr_22198/?gk={gk}"
    r = subprocess.run(["curl", "-sL", "--max-time", "10", "-H", "User-Agent: Mozilla/5.0", test_url],
        capture_output=True, text=True, timeout=15)
    s = BeautifulSoup(r.stdout, "html.parser")
    title = s.title.string.strip() if s.title else "?"
    links = len(s.find_all("a", href=re.compile(r"(content|detail|show|doc_|art)")))
    print(f"  gk={gk}: {len(r.stdout)}b title={title[:40]} content_links={links}")
