#!/usr/bin/env python3
"""凉山州生态环境局 - 环评审批"""
import requests, re, sqlite3, os, sys
from bs4 import BeautifulSoup
from datetime import datetime

BASE = "https://sthj.lsz.gov.cn/zfxxgk/fdzdgknr/hpsp/hpsp_21047"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
DB = "/root/search.db"
site_name = "sthj_lsz_hpsp"

seen_urls = set()
count = 0
max_pages = 5  # daily

def resolve_url(href):
    if href.startswith("http"):
        return href
    href = href.lstrip("./")
    return f"{BASE}/{href}"

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        content_div = (
            soup.select_one("div.trs_editor_view")
            or soup.select_one("div.xilan_word")
            or soup.select_one("div.cont")
        )
        content = ""
        attachments = []
        if content_div:
            content = str(content_div)
        else:
            body = soup.find("body")
            if body:
                content = str(body)
        for a in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", re.I)):
            href = a.get("href", "")
            if href:
                attachments.append(resolve_url(href))
        return content.strip(), attachments
    except Exception as e:
        print(f"  [WARN] detail error: {e}")
        return "", []

conn = sqlite3.connect(DB)
c = conn.cursor()
# Ensure FTS table exists
c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search_v3 USING fts5(title, content, source_url, publish_date, site_name, tokenize='trigram')")

for page in range(max_pages):
    page_url = f"{BASE}/index.html" if page == 0 else f"{BASE}/index_{page}.html"
    print(f"\n=== Page {page+1}: {page_url} ===")
    try:
        r = requests.get(page_url, headers=HEADERS, verify=False, timeout=15)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] page request failed: {e}")
        continue

    soup = BeautifulSoup(r.text, "html.parser")
    items = soup.select("ul.list li")
    if not items:
        print(f"  No items found on page {page+1}")
        continue

    for li in items:
        a = li.find("a")
        span = li.find("span")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title") or a.get_text(strip=True)
        pub_date = span.get_text(strip=True)
        url = resolve_url(href)
        if url in seen_urls:
            continue
        seen_urls.add(url)

        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            print(f"  [SKIP] {title[:40]}... (exists)")
            continue

        content, attachments = get_detail(url)
        if not content or len(content) < 50:
            print(f"  [SKIP] {title[:40]}... (empty content)")
            continue

        summary = re.sub(r"<[^>]+>", "", content)
        summary = re.sub(r"\s+", " ", summary).strip()[:200]

        c.execute(
            "INSERT INTO gov_raw (title, summary, content, page_url, publish_date, category, site_name) VALUES (?,?,?,?,?,?,?)",
            (title.strip(), summary, content, url, pub_date, "环评审批", site_name),
        )
        # FTS sync
        c.execute("SELECT COUNT(*) FROM gov_search_v3 WHERE source_url=?", (url,))
        if c.fetchone()[0] == 0:
            c.execute("INSERT INTO gov_search_v3(title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                      (title.strip(), content, url, pub_date, site_name))
        conn.commit()
        count += 1
        print(f"  [{count}] {title[:50]} ({pub_date})")

conn.close()
print(f"\n===== DONE: {count} new records (FTS synced) =====")
