#!/usr/bin/env python3
"""Debug dataproxy - examine actual embedded data and try correct params."""
import requests, re

headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
BASE = "http://www.jxlc.gov.cn"
DATA_PROXY = BASE + "/module/web/jpage/dataproxy.jsp"

# Get the col1384 page to see the embedded data
r = requests.get("http://www.jxlc.gov.cn/col/col1384/index.html", 
                 timeout=30, headers=headers, verify=False)
html = r.text

# Find the dataStore content (script children of param element)
# Look for the element with ID 70893
m = re.search(r'id=["\']70893["\'][^>]*>(.*?)</(?:div|ul)>', html, re.DOTALL)
if m:
    content = m.group(1)
    print(f"DataStore content ({len(content)} chars):")
    print(content[:500])
else:
    # Try finding the script inside 70893
    m2 = re.search(r'70893.*?<script[^>]*>(.*?)</script>', html, re.DOTALL)
    if m2:
        content = m2.group(1)
        print(f"Script content in 70893 ({len(content)} chars):")
        print(content[:500])
    else:
        print("No dataStore found for 70893")

# Also look at the raw proxyUrl call format
print("\n=== Testing different dataproxy formats ===")
# Test 1: GET request
r2 = requests.get(f"{DATA_PROXY}?col=1&appid=1&webid=5&path=/&columnid=1384&sourceContentType=1&unitid=70893&keyvalue=&pageindex=1&pagenum=15",
                  headers=headers, timeout=30, verify=False)
print(f"GET: {len(r2.text)} bytes: {r2.text[:100]}")

# Test 2: check charset
r3 = requests.post(DATA_PROXY, data={
    "col": "1", "appid": "1", "webid": "5", "path": "/",
    "columnid": "1384", "sourceContentType": "1", "unitid": "70893", 
    "keyvalue": "", "pageindex": "1", "pagenum": "15"
}, headers=headers, timeout=30, verify=False)
r3.encoding = "utf-8"
print(f"POST utf8: {len(r3.text)} bytes: {r3.text[:200]}")

# Test 3: try with different pagenum  
for pn in [10, 15, 18, 20]:
    r4 = requests.post(DATA_PROXY, data={
        "col": "1", "appid": "1", "webid": "5", "path": "/",
        "columnid": "1384", "sourceContentType": "1", "unitid": "70893",
        "keyvalue": "", "pageindex": "1", "pagenum": str(pn)
    }, headers=headers, timeout=30, verify=False)
    r4.encoding = "utf-8"
    print(f"POST pagenum={pn}: {len(r4.text)} bytes: {r4.text[:80]}")
