#!/usr/bin/env python3
import requests
from bs4 import BeautifulSoup
import re

session = requests.Session()
session.headers.update({"User-Agent": "Mozilla/5.0"})

url = "https://www.hx.gov.cn/xxgk/opennessContent/?branch_id=57a3df762c262ea9a00aad3b&column_code=30200"
session.get(url, timeout=15)
ajax_url = "https://www.hx.gov.cn/xxgk/opennessTarget/?branch_id=57a3df762c262ea9a00aad3b&column_code=30200"
session.get(ajax_url, headers={"X-Requested-With": "XMLHttpRequest", "Referer": url}, timeout=15)
session.cookies.set("token_verified", "true", domain="www.hx.gov.cn", path="/")

# Get list
r = session.get(ajax_url, headers={"X-Requested-With": "XMLHttpRequest", "Referer": url}, timeout=15)
r.encoding = "utf-8"
soup = BeautifulSoup(r.text, "html.parser")

# Find the "安徽鹿文" article
rows = soup.select("table tr")
for row in rows:
    tds = row.find_all("td")
    if len(tds) >= 3:
        a = tds[1].find("a")
        if a and "鹿文" in a.get_text():
            href = a.get("href", "")
            title = a.get_text(strip=True)
            date = tds[2].get_text(strip=True)
            print("Found: %s" % title)
            print("URL: %s" % href)
            print("Date: %s" % date)
            
            if not href.startswith("http"):
                href = "https://www.hx.gov.cn" + href
            
            # Fetch detail
            r2 = session.get(href, timeout=15)
            r2.encoding = "utf-8"
            soup2 = BeautifulSoup(r2.text, "html.parser")
            
            print("\n=== Detail page ===")
            # Check if page exists
            if "撤稿" in r2.text or "删除" in r2.text:
                print("Page might be removed!")
            else:
                # Show all div classes
                for div in soup2.find_all("div"):
                    cls = div.get("class", [])
                    if cls:
                        txt = div.get_text(strip=True)[:80]
                        print("  div.%s: %s" % (".".join(cls), txt))
                
                # Find right sidebar with links
                print("\n=== Looking for 意见征集 links ===")
                for el in soup2.find_all(string=re.compile(r"意见征集|征集")):
                    parent = el.parent
                    if parent:
                        print("Parent: <%s class='%s'>: %s" % (parent.name, " ".join(parent.get("class", [])), parent.get_text(strip=True)[:100]))
                        link = parent.find("a") if parent.name != "a" else parent
                        if link and link.name == "a":
                            print("Link: %s -> %s" % (link.get_text(strip=True)[:50], link.get("href", "")))
            
            break
