#!/usr/bin/env python3
"""鸡西市恒山区-建设实施情况 爬虫 (UCMS API)
URL: https://www.jixihengshan.gov.cn/hsq/2db4cac7add84a8aaf9d3ea4368547cc/zfxxgk.shtml
API: GET /common/search/{channelId}
"""
import requests
import sqlite3
import sys
import re
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
BASE_URL = 'https://www.jixihengshan.gov.cn'
API_URL = 'https://www.jixihengshan.gov.cn/common/search/238068a4fcf34a8883b6337cf40ee86b'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '鸡西市恒山区-建设实施情况'
PAGE_SIZE = 10

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def fetch_page(page):
    url = f'{API_URL}?page={page}&_pageSize={PAGE_SIZE}&_isAgg=true&_isJson=true&_template=index'
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        if resp.status_code != 200:
            return [], 0
        data = resp.json()
        results = data.get('data', {}).get('results', [])
        total = data.get('data', {}).get('total', 0)
        items = []
        for r in results:
            title = r.get('title', '')
            time_str = r.get('publishedTimeStr', '')
            pub_date = time_str[:10] if len(time_str) >= 10 else ''
            page_url = BASE_URL + r.get('url', '')
            content_html = r.get('contentHtml', '') or ''
            manuscript_id = str(r.get('manuscriptId', ''))
            items.append((page_url, title, pub_date, content_html, manuscript_id))
        return items, total
    except Exception as e:
        print(f'[ERROR] API page {page}: {e}', file=sys.stderr)
        return [], 0

def get_detail_content(page_url):
    """对于contentHtml为空的文章，抓取详情页获取正文"""
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return ''
        soup = BeautifulSoup(resp.text, 'html.parser')
        content_el = soup.select_one('#zoomcon')
        return str(content_el) if content_el else ''
    except Exception as e:
        print(f'[ERROR] 详情页: {page_url} - {e}', file=sys.stderr)
        return ''

def main():
    daily_mode = '--daily' in sys.argv
    conn = get_conn()
    cur = conn.cursor()
    total_added = 0
    total_skipped = 0
    total_skip_date = 0

    items, total = fetch_page(1)
    if not items and total == 0:
        print('[ERROR] 无法获取数据')
        return

    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f'总条数: {total}, 总页数: {total_pages}')

    all_items = items
    if not daily_mode:
        for page in range(2, total_pages + 1):
            more_items, _ = fetch_page(page)
            all_items.extend(more_items)
            print(f'[API] 第{page}页 -> {len(more_items)} 条')

    for page_url, title, pub_date, content_html, manuscript_id in all_items:
        if pub_date < THREE_YEARS_AGO:
            total_skip_date += 1
            continue

        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
        if cur.fetchone():
            total_skipped += 1
            continue

        # 如果API返回的正文为空，抓详情页
        if not content_html or content_html.strip() == '':
            content_html = get_detail_content(page_url)

        summary = BeautifulSoup(content_html or '', 'html.parser').get_text(strip=True)[:500] if content_html else ''

        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
            (SITE_NAME, page_url, title, content_html, pub_date, summary)
        )
        total_added += 1
        label = f'[{total_added}]' if not daily_mode else '  [ADD]'
        print(f'  {label} {pub_date} {title[:40]}')

    conn.commit()
    print(f'\n同步 FTS ({total_added} 条新增)...')
    cur.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    conn.commit()
    conn.close()
    print(f'\n=== 完成 ===')
    print(f'新增: {total_added}, 跳过重复: {total_skipped}, 超过3年: {total_skip_date}')

if __name__ == '__main__':
    main()
