import json, re, sys

with open('/root/gov_crawler/taixing_api.json') as f:
    data = json.load(f)

html = data['data']['html']

# Find all article links
links = re.findall(r'href="([^"]+)"[^>]*>([^<]+)</a>', html)
for href, text in links[:20]:
    if '/zwgk/xxgk/zdly/sthj/' in href and text.strip():
        print(f'{href} -> {text.strip()[:80]}')

# Get pagination HTML
start = html.find('laypage')
if start > 0:
    print('Pagination HTML:')
    print(html[start:start+600])
