import re

with open('/root/gov_crawler/taixing_detail.html') as f:
    html = f.read()

# Find the main content
con_main = re.search(r'<div class="con-main clear(?:fix)?">(.*?)</div>\s*<!--', html, re.DOTALL)
if con_main:
    content = con_main.group(1)
    print('Content found:')
    print(content[:1500])
else:
    # Try another pattern
    con_main = re.search(r'<div class="con-main clear(?:fix)?">(.*?)</div>', html, re.DOTALL)
    if con_main:
        content = con_main.group(1)
        print('Content found (no comment anchor):')
        print(content[:1500])
    else:
        # Look for con-main
        m = re.search(r'con-main', html)
        if m:
            print('con-main found at:', m.start())
            print(html[m.start():m.start()+2000])
        else:
            print('No con-main found')
            # Print around the article title
            m2 = re.search(r'con-title', html)
            if m2:
                print('Around con-title:')
                print(html[m2.start():m2.start()+2000])
