import re

with open('/root/gov_crawler/taixing_detail.html') as f:
    html = f.read()

# Find all script tags with unitbuild
scripts = re.findall(r'<script[^>]*src="[^"]*unitbuild[^"]*"[^>]*></script>', html)
for s in scripts:
    print('Unitbuild script:', s[:300])

# Also look for build/unit API calls in scripts
api_calls = re.findall(r'build/unit[^"]*', html)
for a in api_calls:
    print('API call:', a[:300])

# Find queryData or buildData
qdata = re.findall(r'(?:queryData|buildData)\s*=\s*([^;]+)', html)
for q in qdata[:5]:
    print('queryData:', q[:300])

# Find the content div more thoroughly
print('\n--- Full con-main ---')
m = re.search(r'<div class="con-main clear(?:fix)?">', html)
if m:
    start = m.end()
    # Look for what comes after
    after = html[start:start+3000]
    print(after)
