#!/usr/bin/env python3
"""Analyze Anji Gov site"""
import requests, re

url = 'https://www.anji.gov.cn/xxgk/bmxxgk/ssthjjajfj/fdzdgknr/zdlyxxgk/shgysy/hjbh/hjyxpj/index.html'
r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=10)
r.encoding = 'utf-8'
h = r.text
print(f'Size: {len(h)}')
t = re.search(r'<title>(.*?)<', h)
if t: print(f'Title: {t.group(1)}')

# CMS markers
for m in ['大汉', 'TRS', 'PowerCMS', 'layui', 'jpage', 'JPage', 'vsb_content', 'zoom', 'articleCon', 'xxgk_content', 'barrierfree', 'contentbox']:
    if m in h:
        print(f'CMS: {m} at pos {h.find(m)}')

# Look for script tags with data
scripts = re.findall(r'<script[^>]*>(.*?)</script>', h, re.DOTALL)
for s in scripts:
    for m in re.finditer(r'ajax|fetch|api|pageSize|totalRecord|getList|dataUrl|listUrl', s, re.I):
        ctx = s[max(0,m.start()-30):m.start()+100]
        print(f'Script AJAX: {ctx}')

# Article links
lis = re.findall(r'<li[^>]*>(.*?)</li>', h, re.DOTALL)
print(f'LI count: {len(lis)}')
if lis:
    for li in lis[:3]:
        am = re.search(r'href="([^"]+)"[^>]*>(.*?)</a>', li)
        if am:
            title = re.sub(r'<[^>]+>', '', am.group(2)).strip()
            print(f'  {title[:40]} -> {am.group(1)[:60]}')

# Div classes
classes = re.findall(r'<div[^>]*class="([^"]*)"', h)
unique = list(set(classes))[:15]
print(f'Div classes: {unique}')

# Check for pagination links
for m in re.finditer(r'href="[^"]*page|index[^"]*"[^>]*>(\d+|\u4e0a\u4e00\u9875|\u4e0b\u4e00\u9875)', h):
    ctx = h[m.start():m.start()+80]
    print(f'Pagination: {ctx}')

# Look for script src
scripts_ext = re.findall(r'<script[^>]*src="([^"]+)"', h)
for s in scripts_ext[:5]:
    print(f'External script: {s}')
