#!/usr/bin/env python3
"""Test ahxx crawler - debug why 0 items"""
import requests, re, sys

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
})

# Test first site
url1 = "https://www.ahxx.gov.cn/public/column/263?type=4&catId=24260769&action=list"
print(f"Fetching: {url1}")
r = session.get(url1, verify=False, timeout=30)
r.encoding = "utf-8"
html = r.text
print(f"Status: {r.status_code}, Size: {len(html)}")

# Column 263
col = "263"
pattern = rf'href="(/public/{col}/(\d+)\.html)"'
matches = re.findall(pattern, html)
print(f"Column {col}: {len(matches)} matches")
for href, aid in matches[:5]:
    print(f"  Article {aid}: {href}")

# Test second site
url2 = "https://www.ahxx.gov.cn/public/column/239?type=4&catId=3846066&action=list"
print(f"\nFetching: {url2}")
r = session.get(url2, verify=False, timeout=30)
r.encoding = "utf-8"
html = r.text
print(f"Status: {r.status_code}, Size: {len(html)}")

col = "239"
pattern = rf'href="(/public/{col}/(\d+)\.html)"'
matches = re.findall(pattern, html)
print(f"Column {col}: {len(matches)} matches")
for href, aid in matches[:5]:
    print(f"  Article {aid}: {href}")
