#!/usr/bin/env python3
"""Check nanpu detail page structure"""
import requests
from bs4 import BeautifulSoup

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
s = requests.Session()
s.headers.update(headers)

# Fetch a recent detail page - from the list
url = 'https://www.nanpu.gov.cn/news/18/'
r = s.get(url, timeout=30)
r.encoding = 'utf-8'
soup = BeautifulSoup(r.text, 'html.parser')

# Find the first news link
for a in soup.find_all('a', href=True):
    href = a['href']
    if '/news_detail/' in href or '/zwxx/' in href:
        full_url = href if href.startswith('http') else f'https://www.nanpu.gov.cn{href}'
        print(f'Fetching detail: {full_url}')
        r2 = s.get(full_url, timeout=30)
        r2.encoding = 'utf-8'
        soup2 = BeautifulSoup(r2.text, 'html.parser')
        
        print(f'Title: {soup2.title.get_text(strip=True) if soup2.title else "N/A"}')
        print(f'Size: {len(r2.text)}')
        
        # Check for rich text containers
        for cls in ['e_richText-11', 'rich_text', 'content', 'article', 'main', 'detail', 'text']:
            el = soup2.find(class_=cls)
            if el:
                print(f'Found .{cls}: {len(el.get_text(strip=True))} chars')
                txt = el.get_text(strip=True)[:100]
                print(f'  text: {txt}')
        
        # Also check by id
        for id_val in ['zoom', 'content', 'article', 'main']:
            el = soup2.find(id=id_val)
            if el:
                print(f'Found #{id_val}: {len(el.get_text(strip=True))} chars')
        
        # Dump all classes with significant text
        print('\nAll classes with text > 50 chars:')
        seen = set()
        for el in soup2.find_all(class_=True):
            txt = el.get_text(strip=True)
            cls_str = ' '.join(el.get('class', []))
            if len(txt) > 50 and cls_str not in seen:
                seen.add(cls_str)
                print(f'  .{cls_str}: {len(txt)} chars')
        
        break
