#!/usr/bin/env python3
"""Check HTML structure of Yangzhou JPAAS response"""
import requests, re, json

url = 'https://kfq.yangzhou.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit'
params = {
    'parseType': 'bulidstatic',
    'webId': 'l3juUa1slUgnLQOQgtJwy',
    'tplSetId': 'CU3LRJa5b4SPXybhicnZl',
    'pageType': 'column',
    'tagId': '列表列表',
    'editType': 'null',
    'pageId': 'Lw8wtupr8vbNtGQMW3yCr',
}

r = requests.get(url, params=params, headers={'User-Agent': 'Mozilla/5.0'}, timeout=10, verify=False)
d = r.json()
html = d['data']['html']

print(f"Full HTML:\n{html}\n")
print("="*60)

# Check raw structure
for li in re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL):
    print(f"LI raw: {li[:200]}\n")

# Try different patterns
for li in re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL):
    # Simple a tag
    am = re.search(r'<a[^>]*>', li)
    if am:
        print(f"A tag: {am.group(0)[:100]}")
    # Href
    hm = re.search(r'href="([^"]*)"', li)
    if hm:
        print(f"Href: {hm.group(1)}")
    # Title attr
    tm = re.search(r'title="([^"]*)"', li)
    if tm:
        print(f"Title attr: {tm.group(1)}")
    # Date span
    sm = re.search(r'<span[^>]*>([^<]*)</span>', li)
    if sm:
        print(f"Span text: {sm.group(1)}")
