#!/usr/bin/env python3
"""Check pagination and detail page for Yangzhou JPAAS"""
import requests, re, json

headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
api_url = 'https://kfq.yangzhou.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit'

base = {
    'parseType': 'bulidstatic',
    'webId': 'l3juUa1slUgnLQOQgtJwy',
    'tplSetId': 'CU3LRJa5b4SPXybhicnZl',
    'pageType': 'column',
    'tagId': '列表列表',
    'editType': 'null',
    'pageId': 'Lw8wtupr8vbNtGQMW3yCr',
}

# Test pageNo=2
params = dict(base)
params['pageNo'] = '2'
r = requests.get(api_url, params=params, headers=headers, timeout=10, verify=False)
d = r.json()
html = d['data']['html']
lis = re.findall(r'<li[^>]*>', html)
print(f'pageNo=2: {len(lis)} lis')

# Check if page 2 has different content from page 1
r1 = requests.get(api_url, params=base, headers=headers, timeout=10, verify=False)
html1 = r1.json()['data']['html']
if html == html1:
    print('Same as page 1 - pageNo not working')
else:
    print('Different from page 1 - pageNo works!')
    # Get date range
    dates = re.findall(r'<span[^>]*>(\d{4}-\d{2}-\d{2})', html)
    if dates:
        print(f'  Page 2 dates: {dates[0]} to {dates[-1]}')

# Try page=2 (without "No")
params2 = dict(base)
params2['page'] = '2'
r2 = requests.get(api_url, params=params2, headers=headers, timeout=10, verify=False)
d2 = r2.json()
html2 = d2['data']['html']
lis2 = re.findall(r'<li[^>]*>', html2)
print(f'page=2: {len(lis2)} lis')

# Detail page analysis
detail_url = 'https://kfq.yangzhou.gov.cn/zfxxgk/fdzdgknr/tzgg/art/2026/art_8be3b1e842544ba8a046769d0bfd0689.html'
rd = requests.get(detail_url, headers=headers, timeout=10, verify=False)
rd.encoding = 'utf-8'
dh = rd.text
print(f'\nDetail page: {rd.status_code}, {len(dh)} chars')

# Look for content div
for c in ['art_con', 'content', 'article', 'zoom', 'xxgk_content', 'xxgk_con', 'main', 'con_detail']:
    if c in dh:
        for m in re.finditer(c, dh):
            ctx = dh[max(0,m.start()-20):m.start()+60]
            print(f'  Found "{c}": {ctx}')

# Check meta for title
tm = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', dh)
if tm: print(f'ArticleTitle: {tm.group(1)}')
tm2 = re.search(r'<title>(.*?)<', dh)
if tm2: print(f'Title: {tm2.group(1)}')
pm = re.search(r'PubDate[^>]*content="([^"]*)"', dh)
if pm: print(f'PubDate: {pm.group(1)}')
