#!/usr/bin/env python3
"""Check detail page meta and encoding"""
import requests, re
from bs4 import BeautifulSoup

headers = {'User-Agent': 'Mozilla/5.0'}
base = 'https://www.xiangfuqu.gov.cn'

url = f'{base}/kfsxfqwz/c00511/pc/content/content_2044599189580066816.html'
r = requests.get(url, headers=headers, timeout=30)
# Try different encodings
print(f'Content-Type: {r.headers.get("Content-Type","")}')
print(f'Apparent encoding: {r.apparent_encoding}')
r.encoding = 'utf-8'
soup = BeautifulSoup(r.text, 'html.parser')
print(f'Title: {soup.title.get_text(strip=True)[:80] if soup.title else "N/A"}')

# Meta
for meta in soup.find_all('meta'):
    name = meta.get('name', '')
    if name:
        content = meta.get('content', '')[:60]
        print(f'Meta {name}: {content}')

# Content area
ac = soup.find('div', class_='article-content')
if ac:
    print(f'\narticle-content: {len(ac.get_text(strip=True))} chars')
    ps = ac.find_all('p')
    print(f'{len(ps)} p-tags')
    for p in ps[:5]:
        txt = p.get_text(strip=True)
        if txt:
            print(f'  <p>: {txt[:120]}')
    
    # Check for tables
    tables = ac.find_all('table')
    print(f'Tables: {len(tables)}')
