#!/usr/bin/env python3
"""爬取临澧县人民政府 - 公告公示（PowerCMS）"""
import requests
import sqlite3
import re
import sys
import time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE = "https://www.linli.gov.cn"
LIST_URL = BASE + "/zwyw/gggs"
DB = "/root/search.db"
CUTOFF = datetime.now() - timedelta(days=365*3)
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
HEADERS = {
    "User-Agent": UA,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
SITE_NAME = "临澧县-公告公示"
CATEGORY = "县区"
MAX_PAGES = 5

s = requests.Session()
s.headers.update(HEADERS)


def fetch_soup(url, retries=3):
    for attempt in range(retries):
        try:
            r = s.get(url, timeout=30)
            r.encoding = "utf-8"
            return BeautifulSoup(r.text, "html.parser")
        except Exception as e:
            if attempt < retries - 1:
                time.sleep(2)
                continue
            print(f"  [ERROR] {url}: {e}", file=sys.stderr)
            return None


def parse_list(soup, cutoff_date):
    items = []
    for li in soup.select("ul.newsList > li"):
        a = li.find("a")
        date_span = li.find("span", class_="date")
        if not a or not date_span:
            continue
        href = a.get("href", "")
        # skip external links
        if href.startswith("http") and BASE not in href:
            continue
        title = a.get("title", "").strip()
        if not title:
            title = a.get_text(strip=True)
        date_str = date_span.get_text(strip=True)
        try:
            pub_date = datetime.strptime(date_str, "%Y-%m-%d")
        except:
            continue
        if pub_date < cutoff_date:
            return items, True  # stopped=true
        url = href if href.startswith("http") else BASE + href
        items.append((title, url, date_str))
    return items, False


def extract_content(soup):
    con = soup.select_one("div.conTxt")
    if not con:
        return ""
    for tag in con.find_all(["script", "style"]):
        tag.decompose()
    html = str(con)
    html = re.sub(r'\s+', ' ', html).strip()
    return html


def save_to_db(items_data):
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    new_count = 0
    for title, url, date_str, content in items_data:
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if c.fetchone():
            continue
        content_title = title
        summary = content[:300] if content else ""
        now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
        c.execute("""INSERT INTO gov_raw (site_name, category, title, page_url, content, summary, publish_date)
                     VALUES (?, ?, ?, ?, ?, ?, ?)""",
                  (SITE_NAME, CATEGORY, content_title, url, content, summary, date_str))
        new_count += 1
    conn.commit()
    conn.close()
    return new_count


def main():
    print(f"[{SITE_NAME}] 开始爬取（前{MAX_PAGES}页）")
    all_data = []
    stopped = False

    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = f"{LIST_URL}_{page}"
        print(f"  第{page}页: {url}")
        soup = fetch_soup(url)
        if not soup:
            break
        items, stopped = parse_list(soup, CUTOFF)
        print(f"    找到 {len(items)} 条")
        for title, item_url, date_str in items:
            print(f"    详情: {title[:40]}...")
            detail_soup = fetch_soup(item_url)
            if not detail_soup:
                continue
            content = extract_content(detail_soup)
            if not content:
                print(f"      [SKIP] 无正文内容")
                continue
            content_len = len(content)
            print(f"      正文长度: {content_len}")
            all_data.append((title, item_url, date_str, content))
            time.sleep(0.5)
        if stopped:
            print("  遇到超3年旧数据，停止翻页")
            break
        time.sleep(1)

    print(f"\n入库 {len(all_data)} 条...")
    n = save_to_db(all_data)
    print(f"新增入库: {n} 条")

    # FTS 由 search.db 触发器 trg_gov_raw_fts_* 统一维护, 无需手工重建 gov_fts

    print("完成!")


if __name__ == "__main__":
    main()
