#!/usr/bin/env python3
"""Check xiangfuqu - look for SSR list content and try API approaches"""
import requests, re, json, sys
from bs4 import BeautifulSoup
sys.stdout.reconfigure(encoding='utf-8')

base = 'https://www.xiangfuqu.gov.cn'
headers = {'User-Agent': 'Mozilla/5.0'}
r = requests.get(f'{base}/kfsxfqwz/c00511/pc/list.html', headers=headers, timeout=60)
html = r.text
soup = BeautifulSoup(html, 'html.parser')

# Check for any hidden article content
body = soup.get_text()
# Find article-like dates in the text body
dates = re.findall(r'(\d{4}-\d{2}-\d{2})', body)
print(f"Dates found in page: {len(dates)} unique set: {set(dates)}")

# Look for the articles JSON data more carefully
# The articles might be in a JS variable
for script in soup.find_all('script'):
    txt = script.get_text(strip=True)
    if 'snszf' in txt:
        # Extract the JSON
        for m in re.finditer(r'snszf:\s*(\[.*?\])\s*,\s*\w+:', txt, re.DOTALL):
            try:
                data = json.loads(m.group(1))
                print(f"\nFound snszf data: {len(data)} items")
                for item in data[:3]:
                    print(f"  {item.get('title','')[:50]} -> {item.get('relationUrl','')[:60]}")
            except:
                print("snszf data: JSON parse failed")
        break

# Try to access relation URLs for this channel's articles
relation_url = f'{base}/kfsxfqwz/c00511/relation/list.json'
r2 = requests.get(relation_url, headers=headers, timeout=15)
print(f'\nRelation list.json: HTTP {r2.status_code}, {r2.text[:200]}')

# Try accessing a content page using the relation article data
content_url = f'{base}/kfsxfqwz/c00511/pc/content/content_2013787546566324224.html'
r3 = requests.get(content_url, headers=headers, timeout=15)
print(f'\nContent page: HTTP {r3.status_code}')
if r3.status_code == 200:
    soup2 = BeautifulSoup(r3.text, 'html.parser')
    print(f'  Title: {soup2.title.get_text(strip=True)[:60] if soup2.title else "N/A"}')
    print(f'  Size: {len(r3.text)}')
    # Meta
    for meta in soup2.find_all('meta'):
        name = meta.get('name', '')
        if name and ('Date' in name or 'Time' in name or 'title' in name.lower()):
            print(f'  Meta {name}: {meta.get("content","")[:50]}')
    # Content divs  
    for div in soup2.find_all('div'):
        cls = ' '.join(div.get('class', [])) if div.get('class') else ''
        id_ = div.get('id', '')
        txt = div.get_text(strip=True)
        if 100 <= len(txt) <= 10000:
            print(f'  div.{cls}#{id_}: {len(txt)} chars - {txt[:80]}')
