#!/usr/bin/env python3
"""Debug crawl_yangzhou_kfq.py fetch_list function"""
import sys, os
os.chdir('/root/gov_crawler')

exec(open('crawl_yangzhou_kfq.py').read().split('if __name__')[0])

items, total = fetch_list(1)
print(f'Total: {total}')
print(f'Items extracted: {len(items)}')

# Debug: fetch raw and try to parse
import json, re

params = dict(API_PARAMS)
params['paramJson'] = json.dumps({'pageNo': 1, 'pageSize': 15})
r = requests.get(API_URL, params=params, headers=HEADERS, timeout=15, verify=False)
d = r.json()
html = d['data']['html']
print(f'HTML len: {len(html)}')
print(f'First 300 chars:')
print(repr(html[:300]))

lis = re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL)
print(f'\nLI count from direct findall: {len(lis)}')

li_pattern = r'href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>'
for li in lis[:2]:
    m = re.search(li_pattern, li, re.DOTALL)
    if m:
        print(f'MATCH: title="{m.group(2)}" date="{m.group(4)}"')
    else:
        print(f'NO MATCH for: {li[:150]}')
