#!/usr/bin/env python3
"""Test crawl_szhbgz_huan.py first 3 pages"""
import sys
sys.path.insert(0, '/root/gov_crawler')
import crawl_szhbgz_huan as c
import time
import re

conn = c.sqlite3.connect(c.DB_PATH)
cur = conn.cursor()
cur.execute("DELETE FROM gov_raw WHERE site_name=?", (c.SITE_NAME,))
conn.commit()
print("Cleaned")

total = 0
for page in [1, 2, 3]:
    url = c.LIST_URL.format(page=page)
    soup = c.get_soup(url)
    if not soup:
        print(f"Page {page}: FAIL")
        break
    ic = soup.find("div", class_="itemright")
    if not ic:
        print(f"Page {page}: no itemright")
        break
    items = ic.select("li.clearfix")
    print(f"Page {page}: {len(items)} items")
    for item in items:
        a = item.find("a", href=re.compile(r"huan_detail\.aspx"))
        if not a:
            continue
        link = a["href"].strip()
        if not link.startswith("http"):
            link = c.BASE_URL + "/" + link
        cur.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (link,))
        if cur.fetchone():
            continue
        time.sleep(0.5)
        title, pub_date, content = c.extract_detail(link)
        if not title:
            title = a.get_text(strip=True)
        if not content or len(content) < 20:
            content = f"[{title}]({link})"
        cur.execute(
            "INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, script_name) VALUES (?,?,?,?,?,?,?, 'test_szhbgz_3.py')",
            (link, title, c.SITE_NAME, pub_date, content, pub_date.replace("-","") if pub_date else "0", title[:200]),
        )
        total += 1
    time.sleep(0.5)

conn.commit()
conn.close()
print(f"Total: {total}")
