#!/usr/bin/env python3
"""Check col1384 and col1611 for search.jsp compatibility."""
import requests, re

headers = {"User-Agent": "Mozilla/5.0"}
BASE = "http://www.jxlc.gov.cn"

# Check col1384 for infotypeId
r = requests.get("http://www.jxlc.gov.cn/col/col1384/index.html", timeout=30, headers=headers, verify=False)
html = r.text
for m in re.finditer(r"number[=\s]+[\"'\\x27]([^\"'\\x27]+)[\"'\\x27]", html):
    v = m.group(1)
    if "D" in v:
        print(f"col1384 number = {v}")

# Try search.jsp for col1384 without specific infotypeId
api = BASE + "/module/xxgk/search.jsp"
data = "infotypeId=&jdid=5&area=LC0036&divid=div680&vc_title=&vc_number=&currpage=1"
r2 = requests.post(api, data=data,
    headers={"User-Agent": "Mozilla/5.0", "X-Requested-With": "XMLHttpRequest",
             "Referer": "http://www.jxlc.gov.cn/col/col1384/index.html",
             "Content-Type": "application/x-www-form-urlencoded"},
    timeout=30, verify=False)
r2.encoding = "utf-8"
nTotal = re.search(r"nTotalCount.*?value=\"(\d+)\"", r2.text)
items_count = len(re.findall(r"<li>", r2.text))
print(f"search.jsp no infotypeId (area=LC0036): items={items_count}, total={nTotal.group(1) if nTotal else '?'}")

# Try col1611 for infotypeId
r3 = requests.get("http://www.jxlc.gov.cn/col/col1611/index.html", timeout=30, headers=headers, verify=False)
html3 = r3.text
for m in re.finditer(r"number[=\s]+[\"'\\x27]([^\"'\\x27]+)[\"'\\x27]", html3):
    v = m.group(1)
    if "D" in v:
        print(f"col1611 number = {v}")

# Check if col1384 uses the same tree.jsp
for m in re.finditer(r"tree\.jsp[^\"']*", html):
    print(f"col1384 tree.jsp: {m.group()}")

# Check col1611 tree.jsp
for m in re.finditer(r"tree\.jsp[^\"']*", html3):
    print(f"col1611 tree.jsp: {m.group()}")

# Check how many records are embedded in col1384 CDATA
cdata_items = re.findall(r"art_1384_\d+", r.text)
print(f"\ncol1384 art_1384 refs: {len(cdata_items)}")
# Also look for art_1384 in the HTML
for m in re.finditer(r"art_1384_\d+", r.text):
    pass  # just counting
hits = len(re.findall(r"art_1384_\d+", r.text))
print(f"all art_1384_ hits: {hits}")
