#!/usr/bin/env python3
import re, subprocess
from bs4 import BeautifulSoup

url = "http://www.taiqian.gov.cn/channel/list/19256.html"
result = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", url],
    capture_output=True, text=True, timeout=30)
soup = BeautifulSoup(result.stdout, "html.parser")

# Get actual href from first list item
ul = soup.find("ul", class_="news-list")
if ul:
    for li in ul.find_all("li")[:3]:
        a = li.find("a")
        if a and a.get("href"):
            href = a["href"]
            print(f"List href: {href}")
            print(f"List text: {a.get_text(strip=True)[:60]}")
            print()

# Now try a detail page - need to construct proper URL
first_href = ul.find("li").find("a")["href"]
if not first_href.startswith("http"):
    first_href = "http://www.taiqian.gov.cn" + ("/" + first_href.lstrip("/") if not first_href.startswith("/") else first_href)

print(f"Full URL: {first_href}")

result2 = subprocess.run(["curl", "-sL", "--max-time", "15", "-H", "User-Agent: Mozilla/5.0", first_href],
    capture_output=True, text=True, timeout=30)
print(f"Status: {len(result2.stdout)} bytes, title: {re.search(r'<title>(.*?)</title>', result2.stdout)}")
# Check if redirected
m = re.search(r"您访问的页面已撤稿", result2.stdout)
if m:
    print("404 - page deleted")
else:
    print("Content received!")

# Check the response for the actual URL
print(f"First 200 chars: {result2.stdout[:200]}")
