#!/usr/bin/env python3
"""Test crawl_szyq.py - first 3 pages + verify data quality"""
import sys
sys.path.insert(0, '/root/gov_crawler')
import crawl_szyq as c
import time
import re
from bs4 import BeautifulSoup

# Clean old bad data
conn = c.sqlite3.connect(c.DB_PATH)
cur = conn.cursor()
cur.execute("DELETE FROM gov_raw WHERE site_name=? AND page_url LIKE '%szyq.gov.cn%'", (c.SITE_NAME,))
conn.commit()
print("Cleaned old bad data")

total = 0
for page in [1, 2, 3]:
    url = c.LIST_URL.format(page=page)
    print(f"Page {page}...", end=" ", flush=True)
    r = c.session.get(url, timeout=20)
    soup = BeautifulSoup(r.text, "html.parser")
    items = soup.find_all("li", class_=re.compile(r"^(odd|even)$"))
    print(f"{len(items)} items")
    for item in items:
        a = item.find("a")
        if not a:
            continue
        link = a.get("href", "")
        if not link:
            continue
        if not link.startswith("http"):
            link = c.BASE_URL + link if link.startswith("/") else c.BASE_URL + "/" + link
        if c.BASE_URL not in link:
            continue
        cur.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (link,))
        if cur.fetchone():
            continue
        time.sleep(1.5)
        title, pub_date, content = c.extract_detail(link)
        if not title:
            title = a.get("title", "") or a.get_text(strip=True)
        if not content:
            content = f"[{title}]({link})"
        try:
            cur.execute(
                "INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, script_name) VALUES (?,?,?,?,?,?,?, 'test_szyq_temp.py')",
                (link, title, c.SITE_NAME, pub_date, content, pub_date.replace("-", "") if pub_date else "0", title[:200]),
            )
            total += 1
        except Exception as e:
            print(f"  ERROR: {e}")
    time.sleep(1.5)

conn.commit()
conn.close()
print(f"\nTotal added: {total}")

# Verify data quality
conn2 = c.sqlite3.connect(c.DB_PATH)
cur2 = conn2.cursor()
cur2.execute("SELECT title, publish_date, length(content) FROM gov_raw WHERE site_name=? ORDER BY publish_date DESC", (c.SITE_NAME,))
rows = cur2.fetchall()
print(f"\nData verification: {len(rows)} total records")
for i, (t, d, clen) in enumerate(rows[:5], 1):
    print(f"  {i}. [{d}] {t[:50]}... ({clen} chars)")
empty_content = [r for r in rows if r[2] < 20]
if empty_content:
    print(f"\n  WARNING: {len(empty_content)} records with empty content!")
    for t, d, clen in empty_content[:3]:
        print(f"    [{d}] {t[:50]}... ({clen} chars)")
conn2.close()
