#!/usr/bin/env python3
"""Analyze guzhen.gov.cn page structure"""
import re
from bs4 import BeautifulSoup

with open('/root/gov_crawler/guzhen_page.html', 'r', encoding='utf-8') as f:
    html = f.read()

print('Page length: %d' % len(html))
soup = BeautifulSoup(html, 'html.parser')

# Find list area - look for common Epoint patterns
# 1. Look for <ul> with article items
for ul in soup.find_all('ul'):
    lis = ul.find_all('li')
    if len(lis) >= 3:
        # Check if they have article links
        links = [li.find('a') for li in lis if li.find('a')]
        if len(links) >= 3:
            hrefs = [a.get('href', '') for a in links if a]
            # Filter to likely article links
            article_links = [h for h in hrefs if h and not h.startswith('#') and not h.startswith('javascript')]
            if len(article_links) >= 3:
                print('\nUl with %d li: class=%s' % (len(lis), ul.get('class', '')))
                for li in lis[:5]:
                    a = li.find('a')
                    if a:
                        print('  %s -> %s' % (a.text.strip()[:60], a.get('href', '')))

# 2. Search for article links in specific divs
article_pattern = re.compile(r'(/content/article|/zfxxgk/public|\.html|\.shtml)')
for a in soup.find_all('a', href=article_pattern):
    href = a.get('href', '')
    if 'javascript' not in href and '#' not in href:
        print('\nArticle link: %s -> %s' % (a.text.strip()[:60], href))

# 3. Check for pagination
for pat in ['page', 'Page', 'pager', 'pageNo', 'pageSize', '下一页', '总记录']:
    matches = [m.start() for m in re.finditer(re.escape(pat) if len(pat) <= 4 else pat, html, re.I)]
    if matches:
        idx = matches[0]
        print('\nPagination [%s]: %s' % (pat, html[max(0,idx-50):idx+100].replace('\n',' ')[:200]))

# 4. Check for Epoint-specific patterns
for meta in ['epoint', 'lonsun', 'LPS.CHANNEL', 'LPS.COLUMN', 'data-col', 'list-data']:
    if meta in html.lower():
        print('\nEpoint marker [%s] found' % meta)

# 5. Look for data/JSON in script tags
for script in soup.find_all('script'):
    if script.string and ('page' in script.string.lower() or 'list' in script.string.lower() or 'data' in script.string.lower()):
        content = script.string[:500]
        print('\nScript content: %s...' % content.replace('\n', ' ')[:300])
