#!/usr/bin/env python3
"""Deep check nanpu detail page structure"""
import requests
from bs4 import BeautifulSoup

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
s = requests.Session()
s.headers.update(headers)

url = 'https://www.nanpu.gov.cn/zwxx/1252.html'
r = s.get(url, timeout=30)
r.encoding = 'utf-8'
soup = BeautifulSoup(r.text, 'html.parser')

# Check what's inside .main
main = soup.find(class_='main')
if main:
    print("=== div.main children ===")
    for child in main.children:
        if child.name:
            cls = child.get('class', [])
            txt = child.get_text(strip=True)[:100]
            print(f'  <{child.name} class={cls}>: {txt}')
    print()
    
    # Check all rich text elements
    for el in main.find_all(class_=True):
        cls_str = ' '.join(el.get('class', []))
        if 'rich' in cls_str.lower() or 'text' in cls_str.lower():
            print(f'Found potential content: .{cls_str}:')
            print(f'  HTML: {str(el)[:500]}')
            print()

# Check ALL elements with class starting with e_richText
for el in soup.find_all(class_=lambda c: c and 'rich' in ' '.join(c).lower() if c else False):
    cls_str = ' '.join(el.get('class', []))
    txt = el.get_text(strip=True)
    print(f'.{cls_str}: {len(txt)} chars "{txt[:100]}"')

# Check for iframe or other content sources
print('\n=== iframes ===')
for iframe in soup.find_all('iframe'):
    print(f'  src={iframe.get("src","")}')

# Check for script-loaded content
scripts = soup.find_all('script')
for s in scripts[:5]:
    txt = s.get_text(strip=True)
    if len(txt) > 50 and ('content' in txt.lower() or 'html' in txt.lower()):
        print(f'\nScript with content/html: {txt[:300]}')
