import json, re, urllib.request, urllib.parse

base_url = 'https://www.taixing.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit'
headers = {'Referer': 'https://www.taixing.gov.cn/zwgk/xxgk/zdly/sthj/index.html'}

for page in [1, 2, 23]:
    params = {
        'pageType': 'column',
        'tagId': '信息公开内容2',
        'parseType': 'bulidstatic',
        'webId': '569012ebb4b9462a9f6c633227a8ef5b',
        'pageId': 'NaLBGqC9X7yXPzYhbzhoI',
        'tplSetId': 'a62207fcac654afba2ff9d55a7839094',
        'paramJson': json.dumps({"pageNo": page, "pageSize": 15}, ensure_ascii=False)
    }
    url = base_url + '?' + urllib.parse.urlencode(params)
    req = urllib.request.Request(url, headers=headers)
    try:
        resp = urllib.request.urlopen(req, timeout=15)
        data = json.loads(resp.read())
        html = data['data']['html']
        texts = re.findall(r'href="([^"]+)"[^>]*>([^<]+)</a>', html)
        print(f'Page {page}: {len(texts)} links')
        for href, text in texts[:3]:
            print(f'  {href.split("/")[-1]} -> {text[:50]}')
        # Check page no in response
        pagi = re.search(r'pageNo="(\d+)"', html)
        print(f'  response pageNo={pagi.group(1) if pagi else "not found"}')
    except Exception as e:
        print(f'Page {page} failed: {e}')
