#!/usr/bin/env python3
"""宝丰县-通知公告 爬虫 (SiteServer CMS)
URL: https://www.baofeng.gov.cn/channels/16434.html
分页: /channels/16434.html (第1页), /channels/16434_{N}.html (第N页)
每页15条，共10页
"""
import requests
import re
import sqlite3
import sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
BASE_URL = 'https://www.baofeng.gov.cn'
LIST_URL = 'https://www.baofeng.gov.cn/channels/16434.html'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '宝丰县-通知公告'

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def get_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None, None, None
        soup = BeautifulSoup(resp.text, 'html.parser')
        title_el = soup.select_one('h1.title')
        title = title_el.get_text(strip=True) if title_el else ''
        date_el = soup.select_one('.articleInfo')
        pub_date = ''
        if date_el:
            m = re.search(r'(\d{4}-\d{2}-\d{2})', date_el.get_text())
            if m:
                pub_date = m.group(1)
        content_el = soup.select_one('.contentTxt#text') or soup.select_one('.contentTxt')
        content = str(content_el) if content_el else ''
        return title, content, pub_date
    except Exception as e:
        print(f'[ERROR] 详情页失败: {url} - {e}', file=sys.stderr)
        return None, None, None

def crawl_page(page_url):
    articles = []
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            print(f'[WARN] 列表页失败: {page_url} status={resp.status_code}', file=sys.stderr)
            return articles
        soup = BeautifulSoup(resp.text, 'html.parser')
        items = soup.select('.wrap-fr-tet ul li a[href*="/contents/"]')
        for a in items:
            href = a.get('href', '')
            if not href.startswith('http'):
                href = BASE_URL + href
            title = a.get('title', '').strip()
            date_span = a.select_one('span')
            date = date_span.get_text(strip=True) if date_span else ''
            articles.append((href, title, date))
    except Exception as e:
        print(f'[ERROR] 列表页抓取失败: {page_url} - {e}', file=sys.stderr)
    return articles

def get_total_pages():
    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        matches = re.findall(r'href="/channels/16434_(\d+)\.html"', resp.text)
        if matches:
            return max(int(m) for m in matches)
        return 1
    except:
        return 1

def main():
    daily_mode = '--daily' in sys.argv
    total_pages = get_total_pages()
    print(f'总页数: {total_pages}')
    conn = get_conn()
    cur = conn.cursor()
    total_added = 0
    total_skipped = 0
    if daily_mode:
        pages_to_crawl = [LIST_URL]
    else:
        pages_to_crawl = [LIST_URL] + [f'https://www.baofeng.gov.cn/channels/16434_{i}.html' for i in range(2, total_pages + 1)]
    for page_url in pages_to_crawl:
        articles = crawl_page(page_url)
        print(f'[列表页] {page_url} -> {len(articles)} 条')
        for href, title, list_date in articles:
            if list_date < THREE_YEARS_AGO:
                print(f'  [SKIP] {list_date} 超过3年: {title}')
                continue
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if cur.fetchone():
                total_skipped += 1
                continue
            detail_title, content, detail_date = get_detail(href)
            if not detail_title:
                detail_title = title
                detail_date = list_date
                print(f'  [WARN] 使用列表页信息: {title}')
            summary = BeautifulSoup(content or '', 'html.parser').get_text(strip=True)[:500] if content else ''
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                (SITE_NAME, href, detail_title, content, detail_date, summary)
            )
            total_added += 1
            label = f'[{total_added}]' if not daily_mode else '  [ADD]'
            print(f'  {label} {detail_date} {detail_title[:40]}')
    conn.commit()
    # 选择性同步FTS（只新增的）
    print(f'\n同步 FTS ({total_added} 条新增)...')
    cur.execute(
        "INSERT INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    conn.commit()
    conn.close()
    print(f'\n=== 完成 ===')
    print(f'新增: {total_added}, 跳过重复: {total_skipped}')
    if daily_mode:
        print('模式: 日跑（仅第1页）')

if __name__ == '__main__':
    main()
