#!/usr/bin/env python3
"""Analyze Yangzhou JPAAS API"""
import requests, re, json

url = 'https://kfq.yangzhou.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit'
base_params = {
    'parseType': 'bulidstatic',
    'webId': 'l3juUa1slUgnLQOQgtJwy',
    'tplSetId': 'CU3LRJa5b4SPXybhicnZl',
    'pageType': 'column',
    'tagId': '列表列表',
    'editType': 'null',
    'pageId': 'Lw8wtupr8vbNtGQMW3yCr',
}

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}

r = requests.get(url, params=base_params, headers=headers, timeout=10, verify=False)
d = r.json()
html = d['data']['html']
print(f'HTML length: {len(html)}')

# Parse items
items = []
for li in re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL):
    m = re.search(r'href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span[^>]*>([^<]*)</span>', li, re.DOTALL)
    if m:
        items.append({'url': m.group(1), 'title': m.group(2), 'date': m.group(4).strip()})

print(f'Items: {len(items)}')
for it in items[:3]:
    print(f'  {it["title"][:35]} | {it["date"]} | {it["url"][:50]}')
for it in items[-2:]:
    print(f'  {it["title"][:35]} | {it["date"]} | {it["url"][:50]}')

# Check if page param works
print('\n--- Try page=2 ---')
p2 = dict(base_params)
p2['page'] = '2'
r2 = requests.get(url, params=p2, headers=headers, timeout=10, verify=False)
d2 = r2.json()
html2 = d2['data']['html']
items2 = []
for li in re.findall(r'<li[^>]*>(.*?)</li>', html2, re.DOTALL):
    m = re.search(r'href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span[^>]*>([^<]*)</span>', li, re.DOTALL)
    if m:
        items2.append({'url': m.group(1), 'title': m.group(2), 'date': m.group(4).strip()})
print(f'Page 2 items: {len(items2)}')

# Check if same as page 1
if items2 and items2[0]['url'] == items[0]['url']:
    print('PAGE 2 IS SAME AS PAGE 1 - different param needed')
else:
    print('Page 2 is different!')
    for it in items2[:3]:
        print(f'  {it["title"][:35]} | {it["date"]} | {it["url"][:50]}')

# Try with offset/size params
print('\n--- Try pageIndex=1 ---')
p3 = dict(base_params)
p3['pageIndex'] = '1'
r3 = requests.get(url, params=p3, headers=headers, timeout=10, verify=False)
d3 = r3.json()
html3 = d3['data']['html']
items3 = len(re.findall(r'<li[^>]*>', html3))
print(f'With pageIndex=1: {items3} items')

# Try pageSize
print('\n--- Try pageSize=100 ---')
p4 = dict(base_params)
p4['pageSize'] = '100'
r4 = requests.get(url, params=p4, headers=headers, timeout=10, verify=False)
d4 = r4.json()
html4 = d4['data']['html']
items4 = len(re.findall(r'<li[^>]*>', html4))
print(f'With pageSize=100: {items4} items')

# Check art URL detail page
print('\n--- Check detail page ---')
detail_url = 'https://kfq.yangzhou.gov.cn/' + items[0]['url'].lstrip('/')
rd = requests.get(detail_url, headers=headers, timeout=10, verify=False)
rd.encoding = 'utf-8'
dh = rd.text
print(f'Detail: {rd.status_code}, {len(dh)} chars')

# Look for content div
for c in ['art_con', 'content', 'article', 'zoom', 'xxgk_content']:
    if c in dh:
        idx = dh.find(c)
        print(f'  Found \"{c}\" at pos {idx}')
        print(f'  Context: {dh[max(0,idx-20):idx+60]}')
