#!/usr/bin/env python3
"""宜春经济技术开发区-公告公示 爬虫 (数融UCAP CMS API)
URL: http://jkq.yichun.gov.cn/ycsjkq/gggs/pc/list.html
API: POST /queryList
"""
import requests
import json
import sqlite3
import sys
import re
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
API_URL = 'http://jkq.yichun.gov.cn/queryList'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Content-Type': 'application/json',
}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '宜春经济技术开发区-公告公示'
PAGE_SIZE = 50  # 单页拿50条，减少请求次数

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def fetch_page(page):
    payload = {
        "current": page,
        "pageSize": PAGE_SIZE,
        "webSiteCode": ["ycsjkq"],
        "channelId": ["1996823590434611200"]
    }
    try:
        resp = requests.post(API_URL, json=payload, headers=HEADERS, timeout=15)
        if resp.status_code != 200:
            print(f'[WARN] API page {page} status={resp.status_code}', file=sys.stderr)
            return [], 0
        data = resp.json()
        results = data.get('data', {}).get('results', [])
        total = data.get('data', {}).get('total', 0)
        items = []
        for item in results:
            src = item.get('source', {})
            title = src.get('title', '')
            pub_date_full = src.get('pubDate', '')
            pub_date = pub_date_full[:10] if len(pub_date_full) >= 10 else pub_date_full
            content_html = src.get('content', {}).get('content', '') if isinstance(src.get('content'), dict) else ''
            # 从content提取纯文本做summary
            summary = BeautifulSoup(content_html or '', 'html.parser').get_text(strip=True)[:500] if content_html else ''
            # 构建URL
            urls_str = src.get('urls', '{}')
            try:
                urls = json.loads(urls_str)
                page_url = 'http://jkq.yichun.gov.cn' + urls.get('pc', '')
            except:
                page_url = ''
            items.append((page_url, title, pub_date, content_html, summary))
        return items, total
    except Exception as e:
        print(f'[ERROR] API page {page}: {e}', file=sys.stderr)
        return [], 0

def main():
    daily_mode = '--daily' in sys.argv

    conn = get_conn()
    cur = conn.cursor()
    total_added = 0
    total_skipped = 0
    total_skip_date = 0

    # 先取第一页拿总条数
    items, total = fetch_page(1)
    if not items:
        print('[ERROR] 无法获取数据')
        return
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f'总条数: {total}, 总页数: {total_pages}')

    # 处理第1页已获取的数据
    all_items = items

    if not daily_mode:
        # 获取剩余页
        for page in range(2, total_pages + 1):
            more_items, _ = fetch_page(page)
            all_items.extend(more_items)
            print(f'[API] 第{page}页 -> {len(more_items)} 条')

    for page_url, title, pub_date, content_html, summary in all_items:
        if pub_date < THREE_YEARS_AGO:
            total_skip_date += 1
            continue

        if page_url:
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
        else:
            # fallback: 用title+date去重
            cur.execute("SELECT id FROM gov_raw WHERE title = ? AND publish_date = ? AND site_name = ?",
                       (title, pub_date, SITE_NAME))
        if cur.fetchone():
            total_skipped += 1
            continue

        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
            (SITE_NAME, page_url, title, content_html, pub_date, summary)
        )
        total_added += 1
        if daily_mode:
            print(f'  [ADD] {pub_date} {title[:40]}')

    conn.commit()

    print(f'\n同步 FTS ({total_added} 条新增)...')
    cur.execute(
        "INSERT INTO gov_search(rowid, title, site_name, summary, content) "
        "SELECT r.id, r.title, r.site_name, r.summary, r.content "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    conn.commit()
    conn.close()

    print(f'\n=== 完成 ===')
    print(f'新增: {total_added}, 跳过重复: {total_skipped}, 超过3年: {total_skip_date}')
    if daily_mode:
        print('模式: 日跑（仅第1页50条）')

if __name__ == '__main__':
    main()
