#!/usr/bin/env python3
"""Quick catch-up for libo remaining pages - no detail fetching, just title+date"""
import sys, os, json, requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from crawler_lib import push_to_searchdb

BASE_URL = "http://www.libo.gov.cn"
SITE_NAME = "荔波县人民政府"
INDUSTRY = "环境公示"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

# Pages that might not have been fully processed
pages_to_check = list(range(19, 41))

# Get already-inserted URLs
import sqlite3
db = sqlite3.connect("/mnt/data/search.db")
existing = set(r[0] for r in db.execute("SELECT source_url FROM gov_raw WHERE script_name='crawl_libo.py'").fetchall())
db.close()
print(f"已有 {len(existing)} 条记录")

total = 0
for page in pages_to_check:
    url = f"{BASE_URL}/zwgk/xxgkml/zdlyxx/hjbh/index_{page}.html"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
    except:
        print(f"页{page}: 请求失败")
        continue
    
    soup = BeautifulSoup(resp.text, "html.parser")
    items = []
    for row in soup.select("table.bg_tb tbody#idData tr.c"):
        link_td = row.select_one("td.tn4 a")
        if not link_td: continue
        title = link_td.get_text(strip=True)
        href = link_td.get("href", "").strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        if href in existing:
            continue
        date_td = row.select_one("td.tn5")
        pub_date = date_td.get_text(strip=True) if date_td else ""
        items.append({
            "title": title, "url": href, "source_url": href,
            "site_name": SITE_NAME, "pub_date": pub_date,
            "summary": "", "content": "", "industry": INDUSTRY,
        })
    
    if items:
        try:
            push_to_searchdb(items, batch_label=f"libo_q{page}")
            total += len(items)
            print(f"页{page}: 新增 {len(items)} 条")
        except Exception as e:
            print(f"页{page}: 入库失败 {e}")
    else:
        print(f"页{page}: 无新数据")

print(f"\n完成，共新增 {total} 条")
