#!/usr/bin/env python3
"""Analyze rendered page HTML for list content"""
import re

with open('/root/gov_crawler/guzhen_rendered.html', 'r', encoding='utf-8') as f:
    html = f.read()

# Search for article-like patterns
# Epoint CMS typically loads data via AJAX and inserts HTML
# Look for the content area

# 1. Find where the list data would be rendered
# Search for common list patterns
for pattern in ['gk_list', 'xxgk-table', 'public-list', 'data-list', 'list-content', 'article-list', 'info-list']:
    idx = html.find(pattern)
    if idx >= 0:
        print('Found [%s] at %d' % (pattern, idx))
        print(html[max(0,idx-50):idx+200].replace('\n', ' '))
        print()

# 2. Look for the container divs
# Check for the main content area
for div_id in ['ls-list', 'ls_content', 'content_list', 'list-data', 'right-list', 'article']:
    matches = re.findall(r'<div[^>]*id=["\']([^"\']*%s[^"\']*)["\'][^>]*>' % div_id, html, re.I)
    if matches:
        print('Div id=[%s]: %s' % (div_id, matches[:3]))

# 3. Find the specific section for this column
print('\n--- Looking for catId 18193621 ---')
idx = html.find('18193621')
if idx >= 0:
    print('At %d: %s...' % (idx, html[idx:idx+200].replace('\n', ' ')[:300]))

# 4. Search for any links with action=detail
detail_links = re.findall(r'href=[\'"]([^\'"]*action=detail[^\'"]*)[\'"][^>]*>([^<]*)</a>', html)
print('\nDetail links: %d' % len(detail_links))
for url, text in detail_links[:5]:
    print('  %s -> %s' % (text.strip()[:50], url))

# 5. Search for any table-like structure with dates
date_pattern = re.findall(r'>(\d{4}[-/]\d{2}[-/]\d{2})<', html)
print('\nDates in page: %d' % len(date_pattern))
if date_pattern:
    print('Sample:', date_pattern[:10])

# 6. Look for the API response embedded in the page
# Epoint sometimes embeds JSON data
json_like = re.findall(r'\{[^}]*"total"[^}]*\}', html)
print('\nJSON-like objects with total: %d' % len(json_like))
for j in json_like[:3]:
    print('  %s...' % j[:200])

# 7. Check the full body content for any rendered data
body_start = html.find('<body')
body_end = html.find('</body>')
if body_start >= 0 and body_end > 0:
    body = html[body_start:body_end]
    # Remove scripts, styles
    body_clean = re.sub(r'<script[^>]*>.*?</script>', '', body, flags=re.DOTALL)
    body_clean = re.sub(r'<style[^>]*>.*?</style>', '', body_clean, flags=re.DOTALL)
    # Look for remaining links
    links = re.findall(r'<a[^>]+href=[\'"]([^\'"]+)[\'"][^>]*>([^<]*)</a>', body_clean)
    print('\nLinks in body (no script/style): %d' % len(links))
    for url, text in links[:20]:
        if text.strip() and not url.startswith('#') and 'javascript' not in url:
            print('  %s -> %s' % (text.strip()[:40], url[:80]))
