#!/usr/bin/env python3
"""Deep analysis of Anji Gov site"""
import requests, re

url = 'https://www.anji.gov.cn/xxgk/bmxxgk/ssthjjajfj/fdzdgknr/zdlyxxgk/shgysy/hjbh/hjyxpj/index.html'
r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=10)
r.encoding = 'utf-8'
h = r.text

# Find all LI with href containing specific patterns
for m in re.finditer(r'<li[^>]*>(.*?)</li>', h, re.DOTALL):
    li = m.group(1)
    hrefs = re.findall(r'href="([^"]+)"', li)
    for href in hrefs:
        if 'hjyxpj' in href or 'hjyx' in href or 'hjbh' in href or '202' in href or 'art' in href:
            title = re.search(r'title="([^"]*)"', li)
            t = title.group(1) if title else ''
            print(f'  ART: {t[:40]} -> {href[:60]}')

# Look for the u-list div which might contain articles
ul = re.search(r'class="u-list"[^>]*>(.*?)</ul>', h, re.DOTALL)
if ul:
    items = re.findall(r'<li[^>]*>(.*?)</li>', ul.group(1), re.DOTALL)
    print(f'\nu-list items: {len(items)}')
    for it in items[:5]:
        a = re.search(r'href="([^"]+)"[^>]*>(.*?)</a>', it)
        if a:
            print(f'  {a.group(2)[:40]} -> {a.group(1)[:50]}')

# Check if articles are embedded in a data-jsp or script
for m in re.finditer(r'<script[^>]*>(.*?)</script>', h, re.DOTALL):
    s = m.group(1)
    for pat in ['data', 'list', 'article', 'news', 'info', 'load', 'json']:
        if pat in s.lower():
            ctx = s[:200]
            print(f'\nScript with "{pat}": {ctx}')
            break

# Look for the main content area
for c in ['Rcont zn_cont', 'zwxxgk_box', 'zfxxgk_item', 'u-list']:
    idx = h.find(c)
    if idx >= 0:
        ctx = h[max(0,idx-50):idx+200]
        print(f'\nContainer "{c}" at {idx}:')
        print(ctx[:300])

# Check page.js files for pagination
for m in re.finditer(r'page|index|Page|pagination|pageNo|pageIndex', h, re.I):
    ctx = h[max(0,m.start()-20):m.start()+80]
    if 'script' not in ctx and 'api' not in ctx:
        print(f'Pagination: {ctx}')
