#!/usr/bin/env python3
"""Check nanpu detail pages with text content"""
import requests
from bs4 import BeautifulSoup

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
s = requests.Session()
s.headers.update(headers)

# Check a news_detail page
urls = [
    'https://www.nanpu.gov.cn/news_detail/1228.html',
    'https://www.nanpu.gov.cn/zwxx/1251.html',
    'https://www.nanpu.gov.cn/zwxx/1252.html',
]

for url in urls:
    print(f'\n=== {url} ===')
    r = s.get(url, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    print(f'Title: {soup.title.get_text(strip=True) if soup.title else "N/A"}')
    
    # Check e_richText-11
    rt = soup.find(class_='e_richText-11')
    if rt:
        txt = rt.get_text(strip=True)
        print(f'.e_richText-11: {len(txt)} chars "{txt[:100]}"')
        # Show full inner HTML
        all_text = []
        for child in rt.children:
            if child.name:
                t = child.get_text(strip=True)
                if t:
                    all_text.append(t)
                elif child.name in ('img', 'figure'):
                    img = child.find('img') or child
                    src = img.get('src','')
                    alt = img.get('alt','')
                    all_text.append(f'[图片: {alt}]')
                elif child.name == 'link':
                    pass
                elif child.name == 'p':
                    t = child.get_text(strip=True)
                    if t:
                        all_text.append(t)
            elif isinstance(child, str) and child.strip():
                all_text.append(child.strip())
        print(f'  Children: {all_text}')
    
    # Also check if there's any text outside rich text
    body = soup.get_text(separator='\n')
    lines = [l.strip() for l in body.split('\n') if l.strip() and len(l.strip()) > 20]
    print(f'  Other text lines (>20 chars):')
    for line in lines[:10]:
        if '发布时间' not in line and '版权' not in line and '地址' not in line and '电话' not in line:
            print(f'    {line[:100]}')
