#!/usr/bin/env python3
"""柴桑区-通知公告 爬虫 (TRS CMS)
URL: https://www.chaisang.gov.cn/zxzx/gsgg/
分页: index.html (第1页), index_{1..24}.html (第2-25页)
每页20条，共25页
"""
import requests
import re
import sqlite3
import sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
BASE_URL = 'https://www.chaisang.gov.cn'
LIST_DIR = '/zxzx/gsgg/'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '柴桑区-通知公告'
TOTAL_PAGES = 25

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def get_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None, None, None
        soup = BeautifulSoup(resp.text, 'html.parser')
        # 标题
        title_el = soup.select_one('p.mainTitle')
        title = title_el.get_text(strip=True) if title_el else ''
        # 日期
        pub_date = ''
        text = soup.get_text()
        m = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', text)
        if m:
            pub_date = m.group(1)
        # 正文
        content_el = soup.select_one('#content.xl-text')
        content = str(content_el) if content_el else ''
        return title, content, pub_date
    except Exception as e:
        print(f'[ERROR] 详情页失败: {url} - {e}', file=sys.stderr)
        return None, None, None

def crawl_page(page_url):
    articles = []
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return articles
        soup = BeautifulSoup(resp.text, 'html.parser')
        items = soup.select('ul.list_main li a[href*="t2026"]')
        for a in items:
            href = a.get('href', '')
            if not href.startswith('http'):
                href = BASE_URL + href if href.startswith('/') else BASE_URL + '/' + href
            title = a.get('title', '').strip()
            if not title:
                title = a.get_text(strip=True)
            # 日期在同级span中
            li = a.find_parent('li')
            date_span = li.select_one('span') if li else None
            list_date = date_span.get_text(strip=True).replace('.', '-') if date_span else ''
            articles.append((href, title, list_date))
    except Exception as e:
        print(f'[ERROR] 列表页抓取失败: {page_url} - {e}', file=sys.stderr)
    return articles

def main():
    daily_mode = '--daily' in sys.argv
    if daily_mode:
        pages_to_crawl = [BASE_URL + LIST_DIR + 'index.html']
    else:
        pages_to_crawl = [BASE_URL + LIST_DIR + 'index.html']
        for i in range(1, TOTAL_PAGES):
            pages_to_crawl.append(f'{BASE_URL}{LIST_DIR}index_{i}.html')

    conn = get_conn()
    cur = conn.cursor()
    total_added = 0
    total_skipped = 0
    total_skip_date = 0

    for page_url in pages_to_crawl:
        articles = crawl_page(page_url)
        pname = page_url.split('/')[-1]
        print(f'[列表页] {pname} -> {len(articles)} 条')
        for href, title, list_date in articles:
            if list_date < THREE_YEARS_AGO:
                total_skip_date += 1
                continue
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if cur.fetchone():
                total_skipped += 1
                continue
            detail_title, content, detail_date = get_detail(href)
            if not detail_title:
                detail_title = title
                detail_date = list_date
                print(f'  [WARN] 使用列表页信息: {title[:30]}')
            summary = BeautifulSoup(content or '', 'html.parser').get_text(strip=True)[:500] if content else ''
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                (SITE_NAME, href, detail_title, content, detail_date, summary)
            )
            total_added += 1
            label = f'[{total_added}]' if not daily_mode else '  [ADD]'
            print(f'  {label} {detail_date} {detail_title[:40]}')

    conn.commit()
    print(f'\n同步 FTS ({total_added} 条新增)...')
    cur.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    conn.commit()
    conn.close()
    print(f'\n=== 完成 ===')
    print(f'新增: {total_added}, 跳过重复: {total_skipped}, 超过3年: {total_skip_date}')

if __name__ == '__main__':
    main()
