#!/usr/bin/env python3
"""Check detail page and pagination for xiangfuqu"""
import requests, re, sys
from bs4 import BeautifulSoup
sys.stdout.reconfigure(encoding='utf-8')

headers = {'User-Agent': 'Mozilla/5.0'}
base = 'https://www.xiangfuqu.gov.cn'

# Check a content page directly 
content_url = f'{base}/kfsxfqwz/c00511/pc/content/content_2044599189580066816.html'
r = requests.get(content_url, headers=headers, timeout=30)
print(f'Content page: HTTP {r.status_code}, Size: {len(r.text)}')

if r.status_code == 200:
    soup = BeautifulSoup(r.text, 'html.parser')
    print(f'Title: {soup.title.get_text(strip=True)[:60] if soup.title else "N/A"}')
    
    # Meta
    for meta in soup.find_all('meta'):
        name = meta.get('name', '')
        if name and ('Date' in name or 'Time' in name or 'title' in name.lower()):
            print(f'Meta {name}: {meta.get("content","")[:50]}')
    
    # Content divs
    for div in soup.find_all('div'):
        cls = ' '.join(div.get('class', [])) if div.get('class') else ''
        id_ = div.get('id', '')
        txt = div.get_text(strip=True)
        if 100 <= len(txt) <= 10000:
            print(f'div.{cls}#{id_}: {len(txt)} chars - {txt[:100]}')
    
    # Check for article-content or similar
    for sel in ['div.article-content', 'div.content', 'div.article', 'div.main-text', '.news-content']:
        el = soup.select_one(sel)
        if el:
            ps = el.find_all('p')
            print(f'\n{sel}: {len(el.get_text(strip=True))} chars, {len(ps)} p-tags')
            for p in ps[:5]:
                ptxt = p.get_text(strip=True)
                if ptxt:
                    print(f'  <p>: {ptxt[:120]}')
            break

# Also check if page 2 exists
print(f'\n\n--- Pagination check ---')
for pn in [2, 3]:
    list_url = f'{base}/kfsxfqwz/c00511/pc/list_{pn}.html'
    r2 = requests.get(list_url, headers=headers, timeout=30)
    print(f'list_{pn}.html: HTTP {r2.status_code}, Size: {len(r2.text)}')
