#!/usr/bin/env python3
"""Check zhuhai detail page structure"""
import requests, re, sys
from bs4 import BeautifulSoup
sys.stdout.reconfigure(encoding='utf-8')

url = 'https://ssthjj.zhuhai.gov.cn/zxfw/xmgsgg/spqgs/content/post_3927014.html'
headers = {'User-Agent': 'Mozilla/5.0'}
r = requests.get(url, timeout=60, headers=headers)
r.encoding = 'utf-8'
soup = BeautifulSoup(r.text, 'html.parser')

print(f'Size: {len(r.text)}')
print(f'Title: {soup.title.get_text(strip=True)}' if soup.title else 'Title: N/A')

# Meta
for meta in soup.find_all('meta'):
    name = meta.get('name', '')
    if name and ('Date' in name or 'Time' in name or 'title' in name.lower()):
        print(f'Meta {name}: {meta.get("content","")[:50]}')

# Find content divs
for div in soup.find_all('div'):
    cls = ' '.join(div.get('class', [])) if div.get('class') else ''
    id_ = div.get('id', '')
    txt = div.get_text(strip=True)
    if 100 <= len(txt) <= 10000:
        print(f'div.{cls}#{id_}: {len(txt)} chars')

# Find the actual content area - look for common patterns
for selector in ['div.content', 'div.article', 'div.detail', 'div.main', 'div.text', 'div.zw']:
    el = soup.select_one(selector)
    if el:
        print(f'\n{selector}: {len(el.get_text(strip=True))} chars')
        ps = el.find_all('p')
        print(f'  {len(ps)} p-tags')
        for p in ps[:5]:
            ptxt = p.get_text(strip=True)
            if ptxt:
                print(f'  <p>: {ptxt[:150]}')

# Also check with class containing "content"
for div in soup.find_all('div'):
    cls = ' '.join(div.get('class', [])) if div.get('class') else ''
    if 'content' in cls.lower() or 'detail' in cls.lower() or 'article' in cls.lower():
        txt = div.get_text(strip=True)
        if len(txt) > 100:
            ps = div.find_all('p')
            print(f'\ndiv.{cls}: {len(txt)} chars, {len(ps)} p-tags')
            for p in ps[:5]:
                ptxt = p.get_text(strip=True)
                if ptxt:
                    print(f'  <p>: {ptxt[:150]}')
