#!/usr/bin/env python3
"""巴彦淖尔市生态环境局-通知公告 爬虫 (TRS CMS)
URL: http://sthjj.bynr.gov.cn/hjzx/tzgg/
分页: index.html (第1页), index_{1..13}.html (第2-14页)
每页20条，共14页
"""
import requests
import re
import sqlite3
import sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
BASE_URL = 'http://sthjj.bynr.gov.cn'
LIST_DIR = '/hjzx/tzgg/'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '巴彦淖尔市生态环境局-通知公告'

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def get_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None, None, None
        soup = BeautifulSoup(resp.text, 'html.parser')
        # 标题
        title_el = soup.select_one('#detail h4.title')
        title = title_el.get_text(strip=True) if title_el else ''
        # 日期
        subtitle = soup.select_one('#detail p.sub-title')
        pub_date = ''
        if subtitle:
            m = re.search(r'(\d{4}-\d{2}-\d{2})', subtitle.get_text())
            if m:
                pub_date = m.group(1)
        # 正文
        content_el = soup.select_one('.content.b-top')
        content = str(content_el) if content_el else ''
        return title, content, pub_date
    except Exception as e:
        print(f'[ERROR] 详情页失败: {url} - {e}', file=sys.stderr)
        return None, None, None

def crawl_page(page_url):
    articles = []
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            print(f'[WARN] 列表页失败: {page_url} status={resp.status_code}', file=sys.stderr)
            return articles
        soup = BeautifulSoup(resp.text, 'html.parser')
        items = soup.select('.text-list-r a.item')
        for a in items:
            href = a.get('href', '')
            if not href.startswith('http'):
                href = BASE_URL + '/hjzx/tzgg/' + href.lstrip('./')
            # 标题在a标签的文本中（去掉span的日期部分）
            date_span = a.select_one('span.rt')
            date = date_span.get_text(strip=True) if date_span else ''
            # 获取标题（a的text内容去掉日期span的text）
            title = a.get_text(strip=True)
            if date:
                title = title.replace(date, '').strip()
            articles.append((href, title, date))
    except Exception as e:
        print(f'[ERROR] 列表页抓取失败: {page_url} - {e}', file=sys.stderr)
    return articles

def main():
    daily_mode = '--daily' in sys.argv

    if daily_mode:
        pages_to_crawl = [BASE_URL + LIST_DIR + 'index.html']
    else:
        pages_to_crawl = [BASE_URL + LIST_DIR + 'index.html']
        for i in range(1, 14):
            pages_to_crawl.append(f'{BASE_URL}{LIST_DIR}index_{i}.html')

    conn = get_conn()
    cur = conn.cursor()
    total_added = 0
    total_skipped = 0
    total_skip_date = 0

    for page_url in pages_to_crawl:
        articles = crawl_page(page_url)
        print(f'[列表页] {page_url.split("/")[-1]} -> {len(articles)} 条')

        for href, title, list_date in articles:
            if list_date < THREE_YEARS_AGO:
                total_skip_date += 1
                continue

            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if cur.fetchone():
                total_skipped += 1
                continue

            detail_title, content, detail_date = get_detail(href)
            if not detail_title:
                detail_title = title
                detail_date = list_date
                print(f'  [WARN] 使用列表页信息: {title[:30]}')

            summary = BeautifulSoup(content or '', 'html.parser').get_text(strip=True)[:500] if content else ''

            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                (SITE_NAME, href, detail_title, content, detail_date, summary)
            )
            total_added += 1
            label = f'[{total_added}]' if not daily_mode else '  [ADD]'
            print(f'  {label} {detail_date} {detail_title[:40]}')

    conn.commit()

    print(f'\n同步 FTS ({total_added} 条新增)...')
    cur.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    conn.commit()
    conn.close()

    print(f'\n=== 完成 ===')
    print(f'新增: {total_added}, 跳过重复: {total_skipped}, 超过3年: {total_skip_date}')
    if daily_mode:
        print('模式: 日跑（仅第1页）')

if __name__ == '__main__':
    main()
