#!/usr/bin/env python3
"""Analyze lylgkfq gsgg page structure"""
import requests, re, sys
from bs4 import BeautifulSoup
sys.stdout.reconfigure(encoding='utf-8')

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
url = 'http://www.lylgkfq.gov.cn/xw/gsgg.htm'
r = requests.get(url, timeout=30, headers=headers)
r.encoding = 'utf-8'

soup = BeautifulSoup(r.text, 'html.parser')
print(f'Title: {soup.title.get_text(strip=True) if soup.title else "N/A"}')
print(f'Size: {len(r.text)}')

# Find list items - VSB9 usually has <a> links in <li> or <table>
# Look for the main content area
for cls in ['list', 'news_list', 'article_list', 'newsinfo', 'main']:
    div = soup.find('div', class_=cls)
    if div:
        print(f'\ndiv.{cls}: {div.get_text(strip=True)[:100]}')
        # Check links
        for a in div.find_all('a', href=True)[:5]:
            print(f'  <a href="{a["href"]}">{a.get_text(strip=True)[:60]}')

# Find all links that look like content links (not nav, not js)
links = soup.find_all('a', href=True)
content_links = []
for a in links:
    href = a.get('href', '')
    txt = a.get_text(strip=True)
    if txt and not href.startswith('#') and not href.startswith('javascript') and not href.startswith('../cssa'):
        if 'info' in href or 'content' in href or '/xw/' in href or '.htm' in href:
            if len(txt) > 10:
                content_links.append((href, txt))

print(f'\nContent links: {len(content_links)}')
for href, txt in content_links[:10]:
    print(f'  {href[:60]}: {txt[:60]}')

# Look for date spans near links
for a in soup.find_all('a', href=True):
    href = a.get('href', '')
    if 'info' in href or 'content' in href:
        txt = a.get_text(strip=True)
        if len(txt) > 10:
            # Check for sibling/next span with date
            parent = a.parent
            spans = parent.find_all('span')
            for span in spans:
                stxt = span.get_text(strip=True)
                if re.search(r'\d{4}[-/]\d{1,2}[-/]\d{1,2}', stxt):
                    print(f'Date span near "{txt[:40]}": {stxt}')
            break

# Check for pagination
for tag in soup.find_all(['div', 'span', 'a']):
    txt = tag.get_text(strip=True)
    if '下一页' in txt or '尾页' in txt or '共' in txt:
        cls = tag.get('class', [])
        print(f'\nPagination: <{tag.name}> class={cls}: {txt[:150]}')
        for a in tag.find_all('a', href=True):
            print(f'  <a href="{a["href"]}">{a.get_text(strip=True)[:30]}')

# Check total count
body = soup.get_text()
m = re.search(r'共(\d+)条', body)
if m:
    print(f'\nTotal: {m.group(1)}条')
m = re.search(r'共(\d+)页', body)
if m:
    print(f'Total: {m.group(1)}页')
