#!/usr/bin/env python3
import requests, re
from bs4 import BeautifulSoup

url = "https://www.jinjiang.gov.cn/ztzl/lhzt/lyxx/sthj/"
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

r = requests.get(url, timeout=30, headers=headers)
html = r.text
soup = BeautifulSoup(html, 'html.parser')

# Find all <a> tags with href containing article-like patterns
for a in soup.find_all('a', href=True):
    href = a['href']
    text = a.get_text(strip=True)
    if text and len(text) > 10 and ('art' in href.lower() or 'content' in href.lower() or 'info' in href.lower() or 'detail' in href.lower() or 'htm' in href.lower() or 'html' in href.lower() or '/' == href[0]):
        if 'gov' in href or href.startswith('/'):
            print(f"{text[:60]} | {href}")

print("\n=== ALL <a> within main content ===")
# Look for content area
main = soup.find(['div', 'article'], id=re.compile(r'content|main|list|right|news|art', re.I))
if not main:
    main = soup.find(['div', 'article'], class_=re.compile(r'content|main|list|right|news|art', re.I))
if main:
    for a in main.find_all('a', href=True):
        text = a.get_text(strip=True)
        href = a['href']
        if text and len(text) > 5:
            print(f"{text[:60]} | {href}")
else:
    print("No main content area found")
    # dump all divs
    for div in soup.find_all('div'):
        c = div.get('class', '')
        i = div.get('id', '')
        if c or i:
            text = div.get_text(strip=True)[:80]
            print(f"DIV class={c} id={i} | {text}")

print("\n=== List area pattern ===")
for ul in soup.find_all('ul'):
    c = ul.get('class', '')
    i = ul.get('id', '')
    text = ul.get_text(strip=True)[:100]
    print(f"UL class={c} id={i} | {text}")
