#!/usr/bin/env python3
"""Test search.jsp API with different infotypeId values."""
import requests, re, html as html_mod

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "X-Requested-With": "XMLHttpRequest",
    "Referer": "https://www.jxlc.gov.cn/col/col1559/index.html?number=D00002D00004",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8"
}
BASE = "https://www.jxlc.gov.cn"
SEARCH_URL = BASE + "/module/xxgk/search.jsp"

tests = [
    # (infotypeId, jdid, divid, area, description)
    ("D00002D00004", "5", "div1432", "", "col1559-政策文件(existing)"),
    ("D00004D00004D00007", "5", "div1432", "", "col4795-环境保护"),
    ("D00004D00003", "5", "div680", "", "col1432-政府信息公开(guess)"),
    ("", "5", "div1432", "", "col1432(no infotypeId)"),
]

for infotypeId, jdid, divid, area, desc in tests:
    data = f"infotypeId={infotypeId}&jdid={jdid}&area={area}&divid={divid}&vc_title=&vc_number=&currpage=1&vc_filenumber=&vc_all=&texttype=&fbtime="
    try:
        r = requests.post(SEARCH_URL, data=data, headers=headers, timeout=30, verify=False)
        r.encoding = "utf-8"
        html = r.text
        # Count list items
        items = re.findall(r'<li>(.*?)</li>', html, re.DOTALL)
        nTotal = re.search(r'nTotalCount.*?value="(\d+)"', html)
        total = nTotal.group(1) if nTotal else "?"
        print(f"[{desc:40s}] infotypeId={infotypeId:30s} items={len(items):4d} total={total}")
    except Exception as e:
        print(f"[{desc:40s}] ERROR: {e}")

# Also try the simple list URL for col1384 (通知公告) - might be a static list
print("\n--- Testing static page for col1384 ---")
r = requests.get("http://www.jxlc.gov.cn/col/col1384/index.html", 
                 headers={"User-Agent": "Mozilla/5.0"}, timeout=30, verify=False)
# Find links in the page
links = re.findall(r'<a[^>]*href="([^"]+)"[^>]*title="([^"]+)"', r.text)
print(f"Links with title attr: {len(links)}")
for href, title in links[:5]:
    print(f"  {title[:40]:45s} -> {href}")

# Try col1611 too (临川经济开发区) - might need different approach
print("\n--- Testing col1611 static page ---")
r2 = requests.get("http://www.jxlc.gov.cn/col/col1611/index.html",
                  headers={"User-Agent": "Mozilla/5.0"}, timeout=30, verify=False)
links2 = re.findall(r'<a[^>]*href="([^"]+)"[^>]*title="([^"]+)"', r2.text)
print(f"Links with title attr: {len(links2)}")
for href, title in links2[:5]:
    print(f"  {title[:40]:45s} -> {href}")
