import re

with open('/root/gov_crawler/taixing_detail.html') as f:
    html = f.read()

# Find title
title = re.search(r'<title>(.*?)</title>', html)
print('Title:', title.group(1) if title else 'not found')

# Check if JPAAS dynamic content
if 'unitbuild' in html or 'build/unit' in html:
    print('Dynamic content (JPAAS unitbuild)')
    # Find the pageId for detail
    m = re.search(r'pageId[\"\']?\s*[:=]\s*[\"\']([^\"\']+)[\"\']', html)
    if m:
        print('pageId:', m.group(1))

# Look for static content first
for cls in ['article', 'content', 'main', 'text', 'detail', 'TRS_Editor', 'zoom', 'art_content', 'info-content']:
    pattern = r'<div[^>]*class="[^"]*' + cls + '[^"]*"[^>]*>(.*?)</div>'
    m = re.search(pattern, html, re.DOTALL)
    if m:
        txt = re.sub(r'<[^>]+>', ' ', m.group(1)).strip()
        txt = re.sub(r'\s+', ' ', txt)[:200]
        print(f'{cls}: {txt[:150]}')

# Print middle section to understand structure
print('\n--- HTML structure ---')
# Find body content
body = re.search(r'<body.*?</body>', html, re.DOTALL)
if body:
    print(body.group()[:2000])
