import sys
sys.path.insert(0, "/root/gov_crawler")

from crawl_cnjf_hjgs import CnjfHjgsCrawler
from bs4 import BeautifulSoup
import re

c = CnjfHjgsCrawler("cnjf-test4", "www.cnjf.com", "https://www.cnjf.com", sync_delay=0.05)
c.max_pages = 1
c.ensure_site()

html = c.http_get("https://www.cnjf.com/aspx/ch/dutylist.aspx?classid=26")
print(f"HTML: {len(html)} bytes")

soup = BeautifulSoup(html, "html.parser")

# Find ALL links
all_links = soup.find_all("a")
print(f"Total links: {len(all_links)}")

# Check each link
count = 0
for a in all_links:
    href = a.get("href", "")
    if "show.aspx" in href or "dutylist" in href:
        id_m = re.search(r"id=(\d+)", href)
        if id_m:
            print(f"  [{href[:60]}] id={id_m.group(1)} title={a.get_text(strip=True)[:50]}")
            count += 1

print(f"show.aspx links: {count}")
