#!/usr/bin/env python3
"""Test page 0 insertion for crawl_gzzn_tzgg.py"""
import sys, os
os.chdir('/root/gov_crawler')
sys.path.insert(0, '/root/gov_crawler')
os.environ['SEARCH_DB'] = '/root/search.db'

import crawl_gzzn_tzgg as m
import sqlite3
from urllib.parse import urljoin
from bs4 import BeautifulSoup

conn = sqlite3.connect(m.DB_PATH)
c = conn.cursor()

soup = m.get_soup(m.BASE)
items = soup.select('ul.NewsList > li')
new = skip = err = 0
for li in items:
    a = li.find('a')
    span = li.find('span')
    if not a or not span:
        continue
    href = a.get('href', '').strip()
    title = a.get('title', '') or a.get_text(strip=True)
    item_date = m.parse_date(span.get_text(strip=True))
    if not href:
        continue
    href = urljoin(m.BASE, href) if not href.startswith('http') else href
    if item_date and item_date < m.THRESHOLD:
        skip += 1
        continue

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
    if c.fetchone():
        skip += 1
        continue

    det_t, det_d, html, e = m.fetch_detail(href)
    if e:
        err += 1
        continue
    if not html or len(html.strip()) < 50:
        skip += 1
        continue

    final_title = det_t or title
    final_date = det_d or item_date
    if not final_date:
        skip += 1
        continue

    date_str = final_date.strftime('%Y-%m-%d')
    dr = m.date_rank(final_date)
    summary = BeautifulSoup(html, 'html.parser').get_text(strip=True)[:200]

    c.execute("""INSERT OR IGNORE INTO gov_raw 
        (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""",
        (m.SITE_NAME, final_title, href, href, date_str, dr, summary, 'published', m.CATEGORY, html))
    if c.rowcount > 0:
        new += 1
        print('+', final_title[:40], '|', date_str)

conn.commit()
conn.close()
print('Page 0 test: new=%d skip=%d err=%d' % (new, skip, err))
