#!/usr/bin/env python3
"""北票市人民政府 - 通知公告"""
import requests, re, sqlite3, os, sys
from bs4 import BeautifulSoup
from datetime import datetime

BASE = "https://www.bp.gov.cn"
LIST_URL = "/bpszf/sylm/ywdt/tzgg/glist.html"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
}
DB = "/root/search.db"
site_name = "bp_tzgg"

seen_urls = set()
count = 0

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        content_div = soup.select_one("div.sunb-infoinfo") or soup.select_one("div.main-wrap")
        content = ""
        if content_div:
            content = str(content_div)
        else:
            body = soup.find("body")
            if body:
                content = str(body)
        return content.strip()
    except Exception as e:
        print(f"  [WARN] detail error: {e}")
        return ""

conn = sqlite3.connect(DB)
c = conn.cursor()
c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search_v3 USING fts5(title, content, source_url, publish_date, site_name, tokenize='trigram')")

try:
    r = requests.get(BASE + LIST_URL, headers=HEADERS, verify=False, timeout=15)
    r.encoding = "utf-8"
except Exception as e:
    print(f"[ERR] {e}")
    sys.exit(1)

soup = BeautifulSoup(r.text, "html.parser")
items = soup.select("ul li")
for li in items:
    a = li.find("a")
    span = li.find("span")
    if not a or not span:
        continue
    href = a.get("href", "")
    title = a.get_text(strip=True)
    pub_date = span.get_text(strip=True)
    if not href.startswith("http"):
        url = href if href.startswith("http") else (href if href.startswith("/") else "/" + href)
        url = BASE + url if url.startswith("/") else url
    else:
        url = href
    if "bp.gov.cn" not in url and "html" not in url:
        continue
    if url in seen_urls:
        continue
    seen_urls.add(url)

    c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
    if c.fetchone():
        print(f"  [SKIP] {title[:40]}... (exists)")
        continue

    content = get_detail(url)
    if not content or len(content) < 50:
        print(f"  [SKIP] {title[:40]}... (empty content)")
        continue

    summary = re.sub(r"<[^>]+>", "", content)
    summary = re.sub(r"\s+", " ", summary).strip()[:200]

    c.execute(
        "INSERT INTO gov_raw (title, summary, content, page_url, publish_date, category, site_name) VALUES (?,?,?,?,?,?,?)",
        (title.strip(), summary, content, url, pub_date, "通知公告", site_name),
    )
    c.execute("SELECT COUNT(*) FROM gov_search_v3 WHERE source_url=?", (url,))
    if c.fetchone()[0] == 0:
        c.execute("INSERT INTO gov_search_v3(title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                  (title.strip(), content, url, pub_date, site_name))
    conn.commit()
    count += 1
    print(f"  [{count}] {title[:50]} ({pub_date})")

conn.close()
print(f"\n===== DONE: {count} new records =====")
