#!/usr/bin/env python3
"""明溪县人民政府 - 综合通知公告 爬虫
TRS WCM API (fjdzapp/data)，直接JSON数据，不需要爬详情页
只爬前5页（75条），日跑增量模式
"""

import os
import sys
import re
import time
import json
import requests
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "明溪县-综合通知公告"
API_URL = "http://www.fjmx.gov.cn/fjdzapp/data"
TOTAL_PAGES = 5
ITEMS_PER_PAGE = 15

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "application/json, text/plain, */*",
    "Content-Type": "application/json",
    "Referer": "http://www.fjmx.gov.cn/zwgk/gggs/zh/",
}

CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
print(f"[info] 日期过滤: >= {CUTOFF_DATE}")


def convert_date(datestr):
    """Convert '2026.06.25 16:34:00' to '2026-06-25'"""
    try:
        dt = datetime.strptime(datestr.strip(), "%Y.%m.%d %H:%M:%S")
        return dt.strftime("%Y-%m-%d")
    except:
        return datestr[:10].replace('.', '-')


def fetch_page(page):
    """Pagination indexing seems 0-based or 1-based at the API level - use page number"""
    payload = {
        "channelid": 100000,
        "classsql": "chnlid=40146",
        "sortfield": "-docorderpri,-docreltime",
        "prepage": ITEMS_PER_PAGE,
        "page": page,
    }
    try:
        r = requests.post(API_URL, json=payload, headers=HEADERS, timeout=15)
        data = r.json()
        if data.get("error"):
            print(f"  [error] API返回错误: {data}")
            return []
        return data.get("data", [])
    except Exception as e:
        print(f"  [error] API请求失败: {e}")
        return []


def extract_attachments(item):
    """从files字段提取附件信息"""
    atts = []
    for f in item.get("files", []):
        href = f.get("_href", "")
        desc = f.get("appdesc", "")
        if href:
            if not href.startswith("http"):
                href = "http://www.fjmx.gov.cn" + href
            atts.append(f'<a href="{href}">{desc}</a>')
    return "<br>".join(atts) if atts else ""


def main():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    old_skip = 0
    err_count = 0

    for page in range(1, TOTAL_PAGES + 1):
        items = fetch_page(page)
        if not items:
            print(f"[warn] 第{page}页无数据")
            continue

        page_new = 0
        page_skip = 0
        for item in items:
            # Parse date
            raw_date = item.get("docreltime", "")
            date = convert_date(raw_date)
            if date < CUTOFF_DATE:
                old_skip += 1
                continue

            title = item.get("doctitle", "").strip()
            url = item.get("chnldocurl", "")
            content = item.get("content", "").strip()
            atts = extract_attachments(item)

            # Combine content and attachments
            if atts:
                content = content + "\n\n<p><strong>附件：</strong></p>\n" + atts

            if not content:
                content = f"[无正文] {title}"

            # 去重
            c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone()[0] > 0:
                page_skip += 1
                continue

            try:
                c.execute("""
                    INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (url, title, content, SITE_NAME, date, title, "通知公告"))
                if c.rowcount > 0:
                    page_new += 1
                    print(f"  [+] {title[:40]}")
            except Exception as e:
                print(f"  [error] 入库失败: {e}")
                err_count += 1

        conn.commit()
        new_count += page_new
        skip_count += page_skip
        print(f"[page {page}/{TOTAL_PAGES}] +{page_new} new, {page_skip} dup, total: {new_count}")

    conn.close()
    print(f"\n[完成] {SITE_NAME}: 新增 {new_count} 条, 重复跳过 {skip_count}, 超3年跳过 {old_skip}, 错误 {err_count}")


if __name__ == "__main__":
    main()
