#!/usr/bin/env python3
"""Debug regex pattern"""
import json, re, requests

url = 'https://kfq.yangzhou.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit'
params = {
    'parseType': 'bulidstatic',
    'webId': 'l3juUa1slUgnLQOQgtJwy',
    'tplSetId': 'CU3LRJa5b4SPXybhicnZl',
    'pageType': 'column',
    'tagId': '\u5217\u8868\u5217\u8868',
    'editType': 'null',
    'pageId': 'Lw8wtupr8vbNtGQMW3yCr',
    'paramJson': json.dumps({'pageNo': 1, 'pageSize': 15}),
}

r = requests.get(url, params=params, headers={'User-Agent': 'Mozilla/5.0'}, timeout=10, verify=False)
d = r.json()
html = d['data']['html']

lis = re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL)
print(f'LI count: {len(lis)}')

# Test simple patterns
li = lis[0]
print(f'\nFull LI[{len(li)}]: {repr(li)}')

# Pattern 1: simple href + title
m1 = re.search(r'href="([^"]+)"', li)
print(f'Pattern href: {m1.group(1) if m1 else "NO MATCH"}')

m2 = re.search(r'title="([^"]*)"', li)
print(f'Pattern title: {m2.group(1) if m2 else "NO MATCH"}')

m3 = re.search(r'<span[^>]*>(\d{4}-\d{2}-\d{2})</span>', li)
print(f'Pattern span date: {m3.group(1) if m3 else "NO MATCH"}')

# Try the full pattern
pat = r'href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>'
m = re.search(pat, li, re.DOTALL)
print(f'Full pattern: {m.group(2) if m else "NO MATCH"}')

# Try with explicit whitespace
pat2 = r'href="([^"]+)"[^>]*title="([^"]*)"[^>]*>([^<]+)</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>'
m2 = re.search(pat2, li, re.DOTALL)
print(f'Pattern2 ([^<]+): {m2.group(2) if m2 else "NO MATCH"}')

# Try without .*? between > and </a>
pat3 = r'title="([^"]*)"[^>]*>([^<]+)</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>'
m3 = re.search(pat3, li, re.DOTALL)
print(f'Pattern3: {m3.group(1) if m3 else "NO MATCH"}')
