#!/usr/bin/env python3
"""Final attempt at dataproxy - try exact parameter format from nextgroup URL."""
import requests, re
from urllib.parse import unquote, quote, urlencode

BASE = "http://www.jxlc.gov.cn"
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

s = requests.Session()
s.headers.update(headers)
s.verify = False

# Visit the page first
s.get("http://www.jxlc.gov.cn/col/col1384/index.html", timeout=30)

# The nextgroup URL params (from dataStore):
# page=1&appid=1&appid=1&webid=5&path=/&columnid=1384&unitid=70893&webname=临川区人民政府&permissiontype=0
# But the webname is double-URL-encoded in the actual URL

# Try 1: GET with double-encoded webname
url = BASE + "/module/web/jpage/dataproxy.jsp"
params = {
    "page": "1",
    "appid": "1",
    "webid": "5",
    "path": "/",
    "columnid": "1384",
    "unitid": "70893",
    "webname": quote("临川区人民政府"),  # single-encode
    "permissiontype": "0"
}
r = s.get(url, params=params, headers={"Referer": "http://www.jxlc.gov.cn/col/col1384/index.html", "X-Requested-With": "XMLHttpRequest"}, timeout=30)
r.encoding = "utf-8"
print(f"Single-encoded webname: {len(r.text)} bytes")

# Try 2: with double-encoded webname (the exact format from nextgroup)
params2 = {
    "page": "1",
    "appid": "1",
    "webid": "5",
    "path": "/",
    "columnid": "1384",
    "unitid": "70893",
    "webname": quote(quote("临川区人民政府")),  # double-encode
    "permissiontype": "0"
}
r2 = s.get(url, params=params2, headers={"Referer": "http://www.jxlc.gov.cn/col/col1384/index.html", "X-Requested-With": "XMLHttpRequest"}, timeout=30)
r2.encoding = "utf-8"
print(f"Double-encoded webname: {len(r2.text)} bytes, content: {r2.text[:80]}")

# Try 3: the exact URL string from nextgroup (with duplicate appid)
exact_url = BASE + "/module/web/jpage/dataproxy.jsp?page=1&appid=1&appid=1&webid=5&path=/&columnid=1384&unitid=70893&webname=%25E4%25B8%25B4%25E5%25B7%259D%25E5%258C%25BA%25E4%25BA%25BA%25E6%25B0%2591%25E6%2594%25BF%25E5%25BA%259C&permissiontype=0"
r3 = s.get(exact_url, headers={"Referer": "http://www.jxlc.gov.cn/col/col1384/index.html", "X-Requested-With": "XMLHttpRequest"}, timeout=30)
r3.encoding = "utf-8"
print(f"Exact nextgroup URL: {len(r3.text)} bytes, content: {r3.text[:80]}")

# Try 4: Check if dataproxy returns XML
r4 = s.get(exact_url, headers={"Referer": "http://www.jxlc.gov.cn/col/col1384/index.html", "X-Requested-With": "XMLHttpRequest", "Accept": "text/xml, application/xml"}, timeout=30)
r4.encoding = "utf-8"
print(f"XML Accept: {len(r4.text)} bytes: {r4.text[:100]}")

# Try 5: maybe the issue is webid=5 but the page loads from http
# Look at the embedded CDATA more carefully for page 1 data extraction
print("\n=== Extracting page 1 data from embedded CDATA ===")
r5 = s.get("http://www.jxlc.gov.cn/col/col1384/index.html", timeout=30)
html = r5.text

# Extract CDATA blocks
records = re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", html, re.DOTALL)
print(f"CDATA records: {len(records)}")

# Parse each record for art_1384
articles = []
for rec in records:
    m = re.search(r"href='([^']+)'[^>]*title='([^']+)'.*?(\d{4}-\d{2}-\d{2})", rec)
    if m:
        articles.append((m.group(1), m.group(2), m.group(3)))

print(f"Articles extracted from page 1: {len(articles)}")
for a, t, d in articles[:3]:
    print(f"  {t[:45]:47s} {d:15s} -> {a}")

# Check if there's a col1611 approach
print("\n=== col1611 analysis ===")
r6 = s.get("http://www.jxlc.gov.cn/col/col1611/index.html", timeout=30)
html6 = r6.text
# Look for embedded art_1611
arts1611 = re.findall(r"art_1611_\d+", html6)
print(f"art_1611 hits: {len(arts1611)}")
# Also check sourceContentType=3 data (param_46616)
for m in re.finditer(r"46616.*?<recordset>(.*?)</recordset>", html6, re.DOTALL):
    print(f"46616 recordset: {len(m.group(1))} chars")
