#!/usr/bin/env python3
"""Extract article list from col1384 and col4795 inline HTML data."""
import requests, re

headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
BASE = "http://www.jxlc.gov.cn"

# col1384 - find actual article data
r = requests.get("http://www.jxlc.gov.cn/col/col1384/index.html",
                 timeout=30, headers=headers, verify=False)
html = r.text

print("=== col1384: looking for article list data ===")
# The dataStore is a script with XML-like content
m = re.search(r'70893.*?<script[^>]*type="text/xml"[^>]*>(.*?)</script>', html, re.DOTALL)
if m:
    raw = m.group(1)
    # Extract recordset
    rs = re.search(r'<recordset>(.*?)</recordset>', raw, re.DOTALL)
    if rs:
        content = rs.group(1)
        print(f"Recordset content ({len(content)} chars):")
        # Find articles
        arts = re.findall(r'<li[^>]*>.*?<a[^>]*href="([^"]+)"[^>]*title="([^"]+)".*?<b>(.*?)</b>.*?</li>', content, re.DOTALL)
        print(f"Articles found: {len(arts)}")
        for a, t, d in arts[:5]:
            print(f"  {t[:45]:47s} {d.strip():15s} -> {a}")

# Also look for the nextpage URL
nm = re.search(r'nextgroup.*?href="([^"]+)"', html)
if nm:
    url = nm.group(1).replace('&amp;', '&')
    print(f"\nNext page URL: {url}")
    # Try fetching
    r2 = requests.get(url if url.startswith('http') else BASE + url,
                      timeout=30, headers=headers, verify=False)
    r2.encoding = 'utf-8'
    print(f"Next page response: {len(r2.text)} bytes")
    if len(r2.text) > 100:
        # Extract articles from response
        arts2 = re.findall(r'<li[^>]*>.*?<a[^>]*href="([^"]+)"[^>]*title="([^"]+)".*?<b>(.*?)</b>.*?</li>', r2.text, re.DOTALL)
        print(f"Articles in response: {len(arts2)}")
        for a, t, d in arts2[:3]:
            print(f"  {t[:40]:42s} {d.strip():15s} -> {a}")

print(f"\n=== col4795: check if search.jsp with infotypeId=D00004D00004 works ===")
data = "infotypeId=D00004D00004&jdid=5&area=&divid=div1432&vc_title=&vc_number=&currpage=1&vc_filenumber=&vc_all=&texttype=&fbtime="
r3 = requests.post("http://www.jxlc.gov.cn/module/xxgk/search.jsp", data=data,
    headers={
        "User-Agent": "Mozilla/5.0",
        "X-Requested-With": "XMLHttpRequest",
        "Referer": "https://www.jxlc.gov.cn/col/col4795/index.html?number=D00004D00004D00007",
        "Content-Type": "application/x-www-form-urlencoded"
    }, timeout=30, verify=False)
r3.encoding = "utf-8"
html3 = r3.text
items = re.findall(r"href='([^']+)'[^>]*title=\"([^\"]+)\".*?<b>(.*?)</b>", html3, re.DOTALL)
nTotal = re.search(r'nTotalCount.*?value="(\d+)"', html3)
print(f"infotypeId=D00004D00004: items={len(items)}, total={nTotal.group(1) if nTotal else '?'}")
for href, title, d in items[:3]:
    url2 = href if href.startswith('http') else BASE + href
    print(f"  {title[:40]:42s} {d.strip():15s} -> {url2}")
