#!/usr/bin/env python3
"""Check ningguo detail page and pagination"""
import requests
from bs4 import BeautifulSoup
import re

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
s = requests.Session()
s.headers.update(headers)

def get_page(url):
    r = s.get(url, timeout=60, verify=False)
    if 'token_verified=true' in r.text:
        s.cookies.set('token_verified', 'true')
        r = s.get(url, timeout=60, verify=False)
    r.encoding = 'utf-8'
    return r

# Check total pages
list_url = 'https://www.ningguo.gov.cn/XxgkContent/showList/383/22474/page_1.html'
r = get_page(list_url)
soup = BeautifulSoup(r.text, 'html.parser')

# Extract total count from page
body = soup.get_text()
total_match = re.search(r'共(\d+)条', body)
if total_match:
    print(f'Total records: {total_match.group(1)}')
else:
    print('No "共X条" found')

# Check page 10 and page 100 to find max
for pn in [5, 10, 20, 50, 100]:
    url = f'https://www.ningguo.gov.cn/XxgkContent/showList/383/22474/page_{pn}.html'
    r = get_page(url)
    soup2 = BeautifulSoup(r.text, 'html.parser')
    for ul in soup2.find_all('ul'):
        lis = ul.find_all('li', recursive=False)
        if len(lis) == 15:
            txt = lis[0].get_text(strip=True)
            is_valid = '环评' in txt or '通知' in txt or '公示' in txt or '公告' in txt
            print(f'Page {pn}: 200, {len(lis)} items, valid={is_valid}, first="{txt[:50]}"')
            break
    else:
        print(f'Page {pn}: {r.status_code}, no list')

# Fetch a detail page
detail_url = 'https://www.ningguo.gov.cn/OpennessContent/show/3784926.html'
r = get_page(detail_url)
soup = BeautifulSoup(r.text, 'html.parser')
print(f'\nDetail page: {detail_url}')
print(f'Status: {r.status_code}, Size: {len(r.text)}')
print(f'Title: {soup.title.get_text(strip=True) if soup.title else "N/A"}')

# Find main content
for cls_name in ['content', 'article', 'main', 'text', 'xxgk_content', 'detail']:
    div = soup.find('div', class_=cls_name)
    if div:
        txt = div.get_text(strip=True)[:100]
        print(f'  div.{cls_name}: {txt}...')
        # Show structure - paragraphs
        ps = div.find_all('p')
        if ps:
            for p in ps[:5]:
                ptxt = p.get_text(strip=True)[:80]
                if ptxt:
                    print(f'    <p>: {ptxt}')
        # Check tables
        tbls = div.find_all('table')
        print(f'    tables: {len(tbls)}')
        break

# Also check for meta PubDate
for meta in soup.find_all('meta'):
    name = meta.get('name', '')
    if 'Date' in name or 'Time' in name:
        content = meta.get('content', '')
        print(f'  Meta {name}: {content}')

# Check all divs
print('\nAll content divs:')
for div in soup.find_all('div'):
    cls = div.get('class', [])
    id_ = div.get('id', '')
    txt = div.get_text(strip=True)
    if len(txt) > 50 and len(txt) < 5000:
        print(f'  div class={cls} id={id_}: {txt[:80]}...')
