#!/usr/bin/env python3
"""Analyze ningguo list page - detailed"""
import requests
from bs4 import BeautifulSoup
import re

url = 'https://www.ningguo.gov.cn/XxgkContent/showList/383/22474/page_1.html'
headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
s = requests.Session()
s.headers.update(headers)

r = s.get(url, timeout=60, verify=False)
if 'token_verified=true' in r.text:
    s.cookies.set('token_verified', 'true')
    r = s.get(url, timeout=60, verify=False)

# Check encoding
print(f'Apparent encoding: {r.apparent_encoding}')
print(f'Encoding: {r.encoding}')
r.encoding = 'utf-8'

soup = BeautifulSoup(r.text, 'html.parser')
print(f'Title: {soup.title.get_text(strip=True) if soup.title else "N/A"}')

# Find the 15-li ul specifically
uls = soup.find_all('ul')
for ul in uls:
    lis = ul.find_all('li', recursive=False)
    if len(lis) == 15:
        print(f'\nFound main list UL with {len(lis)} items:')
        for li in lis:
            a = li.find('a', href=True)
            href = a['href'] if a else 'N/A'
            txt = li.get_text(strip=True)
            # Extract date if present
            date_match = re.search(r'20\d{2}-\d{2}-\d{2}', txt)
            date = date_match.group(0) if date_match else ''
            # Title = text without date
            title = txt.replace(date, '').strip()
            print(f'  title="{title[:60]}" date={date} href={href}')
        break

# Also look at page 2 for pagination check
print('\n=== Page 2 ===')
url2 = 'https://www.ningguo.gov.cn/XxgkContent/showList/383/22474/page_2.html'
r2 = s.get(url2, timeout=60, verify=False)
r2.encoding = 'utf-8'
soup2 = BeautifulSoup(r2.text, 'html.parser')
for ul in soup2.find_all('ul'):
    lis = ul.find_all('li', recursive=False)
    if len(lis) == 15:
        print(f'Page 2 also has {len(lis)} items')
        for li in lis[:3]:
            print(f'  {li.get_text(strip=True)[:80]}')
        break

# Total pages - search for page info
body = soup.get_text()
total_match = re.search(r'共(\d+)条', body)
page_match = re.search(r'共(\d+)页', body)
if total_match:
    print(f'\nTotal records: {total_match.group(1)}')
if page_match:
    print(f'Total pages: {page_match.group(1)}')
else:
    # Check page_2 to see if it works
    print(f'Page 2 status: {r2.status_code}, size: {len(r2.text)}')
    # Check for pagination links
    pagination = soup.find('div', class_=re.compile(r'page|turn|pagi'))
    if not pagination:
        # Try finding page number text in body
        for tag in soup.find_all(['a', 'span', 'div']):
            t = tag.get_text(strip=True)
            if '1' in t and ('2' in t or '页' in t or '末页' in t):
                print(f'Possible pagination: <{tag.name}> {t[:100]}')
                break
