#!/usr/bin/env python3
"""Try dataproxy with session cookies (visit page first)."""
import requests, re

BASE = "http://www.jxlc.gov.cn"
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

s = requests.Session()
s.headers.update(headers)
s.verify = False

# First visit the page to get cookies/session
r = s.get("http://www.jxlc.gov.cn/col/col1384/index.html", timeout=30)
print(f"Page load: {len(r.text)} bytes, cookies: {dict(s.cookies)}")

# Now try dataproxy
dp_headers = {
    "Referer": "http://www.jxlc.gov.cn/col/col1384/index.html",
    "X-Requested-With": "XMLHttpRequest",
}
url = BASE + "/module/web/jpage/dataproxy.jsp"
params = {
    "page": "1", "appid": "1", "webid": "5", "path": "/",
    "columnid": "1384", "unitid": "70893",
    "webname": "临川区人民政府", "permissiontype": "0"
}
r2 = s.get(url, params=params, headers=dp_headers, timeout=30)
r2.encoding = "utf-8"
print(f"dataproxy GET: {len(r2.text)} bytes")

# Try POST
r3 = s.post(url, data=params, headers=dp_headers, timeout=30)
r3.encoding = "utf-8"
print(f"dataproxy POST: {len(r3.text)} bytes")

# Try with page 2
r4 = s.get(url, params={**params, "page": "2"}, headers=dp_headers, timeout=30)
r4.encoding = "utf-8"
print(f"dataproxy page 2: {len(r4.text)} bytes")

# Try with the exact nextgroup URL format
# The nextgroup URL has double-encoded webname and double appid
next_url = BASE + "/module/web/jpage/dataproxy.jsp?page=1&appid=1&appid=1&webid=5&path=/&columnid=1384&unitid=70893&webname=%25E4%25B8%25B4%25E5%25B7%259D%25E5%258C%25BA%25E4%25BA%25BA%25E6%25B0%2591%25E6%2594%25BF%25E5%25BA%259C&permissiontype=0"
r5 = s.get(next_url, headers=dp_headers, timeout=30)
r5.encoding = "utf-8"
print(f"exact nextgroup: {len(r5.text)} bytes, content: {r5.text[:100]}")

# Try col4795 search.jsp again with session
print(f"\n=== col4795 with session ===")
r6 = s.post("https://www.jxlc.gov.cn/module/xxgk/search.jsp", data={
    "infotypeId": "D00004D00004", "jdid": "5", "area": "", "divid": "div1432",
    "vc_title": "", "vc_number": "", "currpage": "1", "vc_filenumber": "",
    "vc_all": "", "texttype": "", "fbtime": ""
}, headers={
    "X-Requested-With": "XMLHttpRequest",
    "Referer": "https://www.jxlc.gov.cn/col/col4795/index.html?number=D00004D00004D00007",
    "Content-Type": "application/x-www-form-urlencoded"
}, timeout=30)
r6.encoding = "utf-8"
nTotal = re.search(r'nTotalCount.*?value="(\d+)"', r6.text)
items = re.findall(r"href='([^']+)'[^>]*title=\"([^\"]+)\".*?<b>(.*?)</b>", r6.text, re.DOTALL)
print(f"col4795: items={len(items)}, total={nTotal.group(1) if nTotal else '?'}")

# Try with urlencoded data format
r7 = s.post("https://www.jxlc.gov.cn/module/xxgk/search.jsp",
    data="infotypeId=D00004D00004&jdid=5&area=&divid=div1432&vc_title=&vc_number=&currpage=1&vc_filenumber=&vc_all=&texttype=&fbtime=",
    headers={
        "X-Requested-With": "XMLHttpRequest",
        "Referer": "https://www.jxlc.gov.cn/col/col4795/index.html?number=D00004D00004D00007",
        "Content-Type": "application/x-www-form-urlencoded"
    }, timeout=30)
r7.encoding = "utf-8"
nTotal7 = re.search(r'nTotalCount.*?value="(\d+)"', r7.text)
items7 = re.findall(r"href='([^']+)'[^>]*title=\"([^\"]+)\".*?<b>(.*?)</b>", r7.text, re.DOTALL)
print(f"col4795 urlencoded: items={len(items7)}, total={nTotal7.group(1) if nTotal7 else '?'}")
