#!/usr/bin/env python3
"""Test dataproxy.jsp API and extract page 1 content."""
import requests, re

headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

BASE = "http://www.jxlc.gov.cn"
DATA_PROXY = BASE + "/module/web/jpage/dataproxy.jsp"

# col1384 has param_70893, sourceContentType=1, perPage=15, totalRecord=1822
# col4795 has param_52954, sourceContentType=3, perPage=18, totalRecord=18
# col1611_46632: sourceContentType=1, perPage=15, totalRecord=2

tests = [
    {"name": "col1384_70893", "col": 1, "appid": "1", "webid": 5, "path": "/", "columnid": 1384, "sourceContentType": 1, "unitid": 70893, "page": 1},
    {"name": "col4795_52954", "col": 1, "appid": "1", "webid": 5, "path": "/", "columnid": 4795, "sourceContentType": 3, "unitid": 52954, "page": 1},
    {"name": "col1611_46632", "col": 1, "appid": "1", "webid": 5, "path": "/", "columnid": 1611, "sourceContentType": 1, "unitid": 46632, "page": 1},
    {"name": "col1611_46616", "col": 1, "appid": "1", "webid": 5, "path": "/", "columnid": 1611, "sourceContentType": 3, "unitid": 46616, "page": 1},
]

for t in tests:
    data = {
        "col": t["col"], "appid": t["appid"], "webid": t["webid"], "path": t["path"],
        "columnid": t["columnid"], "sourceContentType": t["sourceContentType"],
        "unitid": t["unitid"], "keyvalue": "", "pageindex": t["page"], "pagenum": 15
    }
    try:
        r = requests.post(DATA_PROXY, data=data, headers=headers, timeout=30, verify=False)
        r.encoding = "utf-8"
        html = r.text
        # Check for articles
        arts = re.findall(r'href="([^"]*art_\d+_\d+\.html)"[^>]*>([^<]+)', html)
        dates = re.findall(r'(\d{4}-\d{2}-\d{2})', html)
        total = len(arts)
        print(f"[{t['name']:20s}] items={total}, dates={len(dates)}, len={len(html)}")
        for a in arts[:3]:
            print(f"    {a[1][:45]:47s} -> {a[0]}")
    except Exception as e:
        print(f"[{t['name']:20s}] ERROR: {e}")
