#!/usr/bin/env python3
"""Check ningguo detail page content structure"""
import requests
from bs4 import BeautifulSoup
import re

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
s = requests.Session()
s.headers.update(headers)

def get_page(url):
    r = s.get(url, timeout=60, verify=False)
    if 'token_verified=true' in r.text:
        s.cookies.set('token_verified', 'true')
        r = s.get(url, timeout=60, verify=False)
    r.encoding = 'utf-8'
    return r

# Check first detail page
detail_url = 'https://www.ningguo.gov.cn/OpennessContent/show/3784926.html'
r = get_page(detail_url)
soup = BeautifulSoup(r.text, 'html.parser')

# Content from div#zoom
zoom = soup.find('div', id='zoom')
if zoom:
    print(f'div#zoom: {len(zoom.get_text())} chars')
    ps = zoom.find_all('p')
    print(f'  <p> tags: {len(ps)}')
    for p in ps:
        txt = p.get_text(strip=True)
        if txt:
            print(f'  <p>: {txt[:100]}')
            if len(txt) > 100:
                print(f'    ...{txt[-50:]}')
    
    tables = zoom.find_all('table')
    print(f'  <table> tags: {len(tables)}')
    for i, tbl in enumerate(tables[:2]):
        print(f'  Table#{i}:')
        rows = tbl.find_all('tr')
        for tr in rows[:4]:
            cells = [td.get_text(strip=True)[:25] for td in tr.find_all(['td', 'th'])]
            print(f'    {cells}')
    
    # Show all children tags
    print('\n  Direct children:')
    for child in zoom.children:
        if child.name:
            print(f'    <{child.name}>: {child.get_text(strip=True)[:80]}')
    
    # Convert to our standard format
    paragraphs = []
    for p in zoom.find_all('p'):
        txt = p.get_text(strip=True)
        if txt:
            paragraphs.append(txt)
    if not paragraphs:
        # Simple text extraction
        text = zoom.get_text(separator='\n').strip()
        paragraphs = [l.strip() for l in text.split('\n') if l.strip()]
    
    print(f'\n  Total paragraphs: {len(paragraphs)}')
    for i, p in enumerate(paragraphs):
        print(f'  [{i+1}] {p[:80]}')

# Check another detail page with table content
print('\n\n=== Second detail page ===')
r2 = get_page('https://www.ningguo.gov.cn/OpennessContent/show/3456123.html')
soup2 = BeautifulSoup(r2.text, 'html.parser')
zoom2 = soup2.find('div', id='zoom')
if zoom2:
    ps = zoom2.find_all('p')
    tables = zoom2.find_all('table')
    print(f'  {len(ps)} paragraphs, {len(tables)} tables')
    for p in ps[:5]:
        print(f'  <p>: {p.get_text(strip=True)[:100]}')
    for tbl in tables[:2]:
        rows = tbl.find_all('tr')
        print(f'  <table>: {len(rows)} rows')
        for tr in rows[:4]:
            cells = [td.get_text(strip=True)[:20] for td in tr.find_all(['td', 'th'])]
            print(f'    {cells}')
