#!/usr/bin/env python3
"""Test dataproxy with proper Referer and different content types."""
import requests, re

BASE = "http://www.jxlc.gov.cn"

# Test with proper Referer
headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": "http://www.jxlc.gov.cn/col/col1384/index.html",
}

# Try GET with Referer
r = requests.get(BASE + "/module/web/jpage/dataproxy.jsp", params={
    "page": "1", "appid": "1", "webid": "5", "path": "/",
    "columnid": "1384", "unitid": "70893",
    "webname": "临川区人民政府", "permissiontype": "0"
}, headers=headers, timeout=30, verify=False)
r.encoding = "utf-8"
print(f"GET with Referer: {len(r.text)} bytes")
if len(r.text) > 100:
    print(r.text[:300])

# Try with X-Requested-With
headers2 = dict(headers)
headers2["X-Requested-With"] = "XMLHttpRequest"
r2 = requests.get(BASE + "/module/web/jpage/dataproxy.jsp", params={
    "page": "1", "appid": "1", "webid": "5", "path": "/",
    "columnid": "1384", "unitid": "70893",
    "webname": "临川区人民政府", "permissiontype": "0"
}, headers=headers2, timeout=30, verify=False)
r2.encoding = "utf-8"
print(f"GET with X-Requested-With: {len(r2.text)} bytes")
if len(r2.text) > 100:
    print(r2.text[:300])

# Try the nextgroup URL exactly as it appears in the page
next_page = "/module/web/jpage/dataproxy.jsp?page=1&appid=1&appid=1&webid=5&path=/&columnid=1384&unitid=70893&webname=%25E4%25B8%25B4%25E5%25B7%259D%25E5%258C%25BA%25E4%25BA%25BA%25E6%25B0%2591%25E6%2594%25BF%25E5%25BA%259C&permissiontype=0"
r3 = requests.get(BASE + next_page, headers=headers2, timeout=30, verify=False)
r3.encoding = "utf-8"
print(f"Exact nextgroup URL with AJAX: {len(r3.text)} bytes ({r3.text[:50]})")

# Actually let me check what the 54 empty bytes look like in hex
print(f"\nEmpty response hex: {r3.text.encode('utf-8')[:50]}")
print(f"Empty response repr: {repr(r3.text[:50])}")

# Also try the search.jsp from col4795 again with proper Referer
print(f"\n=== col4795 search.jsp with proper Referer ===")
data = "infotypeId=D00004D00004&jdid=5&area=&divid=div1432&vc_title=&vc_number=&currpage=1&vc_filenumber=&vc_all=&texttype=&fbtime="
r4 = requests.post(BASE + "/module/xxgk/search.jsp", data=data,
    headers={
        "User-Agent": "Mozilla/5.0",
        "X-Requested-With": "XMLHttpRequest",
        "Referer": "https://www.jxlc.gov.cn/col/col4795/index.html?number=D00004D00004D00007",
        "Content-Type": "application/x-www-form-urlencoded"
    }, timeout=30, verify=False)
r4.encoding = "utf-8"
html4 = r4.text
items = re.findall(r"href='([^']+)'[^>]*title=\"([^\"]+)\".*?<b>(.*?)</b>", html4, re.DOTALL)
nTotal = re.search(r'nTotalCount.*?value="(\d+)"', html4)
print(f"infotypeId=D00004D00004: items={len(items)}, total={nTotal.group(1) if nTotal else '?'}")
if items:
    for href, title, d in items[:2]:
        print(f"  {title[:40]:42s} {d.strip():15s}")
