#!/usr/bin/env python3
"""
晋江市人民政府 - 生态环境局 - 绿色转型（生态环境保护）
CMS: Terton (天润) SSP
列表: POST /ssp/search/api/v2/external (批量100条)
详情: 取 detail HTML 正文 (含表格等完整HTML)
"""

import os
import re
import sys
import time
import json
import requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "jinjiang.gov.cn-绿色转型-生态环境保护"
API_URL = "https://www.jinjiang.gov.cn/ssp/search/api/v2/external"
BASE_URL = "https://www.jinjiang.gov.cn"
PAGE_SIZE = 100
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
print(f"Cutoff date: {CUTOFF}")
CUTOFF_TS = int(datetime.strptime(CUTOFF, "%Y-%m-%d").timestamp() * 1000)

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Content-Type": "application/json",
}

session = requests.Session()
session.headers.update(HEADERS)
session_detail = requests.Session()
session_detail.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
})


def fetch_api_page(page):
    """Fetch one page of items from the API."""
    body = {
        "siteId": "000000008a939277018a9bf9d5960057",
        "apiName": "WCMDOCAP",
        "pageSize": PAGE_SIZE,
        "page": page,
        "sortField": "docreltime desc",
        "filter": {
            "docstatus": [{"eq": 10}],
            "chnlid": [{"eq": "50128"}],
        }
    }
    try:
        r = session.post(API_URL, json=body, timeout=30)
        r.encoding = "utf-8"
        return r.json()
    except Exception as e:
        print(f"  [ERR] API page {page}: {e}")
        return None


def fetch_detail(url):
    """Fetch detail page and extract content HTML from TRS_Editor / article_area."""
    try:
        r = session_detail.get(url, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
    except Exception as e:
        print(f"  [ERR] detail fetch: {url[:80]} - {e}")
        return None

    soup = BeautifulSoup(r.text, "html.parser")

    # Try TRS_Editor first (most common for rich content)
    trs = soup.select_one(".TRS_Editor")
    if trs:
        content = str(trs)
        return content

    # Fallback: article_area
    area = soup.select_one(".article_area")
    if area:
        return str(area)

    # Last resort: any content div
    content_div = soup.select_one(".content, #zoom, .g-detailbox")
    if content_div:
        return str(content_div)

    return None


def ts_to_date(ts_ms):
    try:
        ts = int(ts_ms) / 1000
        return datetime.fromtimestamp(ts).strftime("%Y-%m-%d")
    except:
        return ""


def main():
    import sqlite3

    conn = sqlite3.connect(SEARCH_DB)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=15000")
    cur = conn.cursor()

    total_new = 0
    total_skip = 0
    total_dup = 0
    total_fail = 0
    page = 1
    total_pages = 999

    while page <= total_pages:
        print(f"\n--- Page {page}/{total_pages} ---")
        res = fetch_api_page(page)
        if not res or res.get("num") != 200:
            print(f"  [SKIP] page {page} API error")
            break

        data = res.get("data", {})
        total = int(data.get("total", 0))
        page_size = int(data.get("pageSize", PAGE_SIZE))
        total_pages = (total + page_size - 1) // page_size
        result = data.get("result", [])

        if not result:
            break

        print(f"  API: {len(result)} items (total: {total}, pages: {total_pages})")

        page_new = 0
        page_skip = 0
        for item in result:
            docpuburl = item.get("docpuburl", "")
            doctitle = item.get("doctitle", "")
            if not docpuburl or not doctitle:
                continue

            pubdate_ts = item.get("pubdate", 0)
            date_str = ts_to_date(pubdate_ts)

            if date_str and int(pubdate_ts) < CUTOFF_TS:
                page_skip += 1
                total_skip += 1
                continue

            # Fetch detail page for rich HTML content (with tables, styles, etc.)
            content_html = fetch_detail(docpuburl)
            if not content_html:
                print(f"  [WARN] no content for {doctitle[:40]}, using plain text")
                content_html = item.get("doccontent") or ""
                total_fail += 1

            date_rank = int(date_str.replace("-", "")) if date_str else 0

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, doctitle, docpuburl, date_str, content_html, date_rank, 'hjxx')
                )
                if cur.rowcount > 0:
                    page_new += 1
                    total_new += 1
                    if page_new % 10 == 0:
                        conn.commit()
                else:
                    total_dup += 1
            except Exception as e:
                print(f"  [ERR] insert: {e}")

            time.sleep(0.3)  # polite delay for detail page fetch

        conn.commit()
        print(f"  Page {page}: +{page_new} new, {page_skip} skip (total +{total_new})")

        if result and int(result[-1].get("pubdate", 0)) < CUTOFF_TS:
            if page_skip > len(result) / 2:
                unskipped = [r for r in result if int(r.get("pubdate", 0)) >= CUTOFF_TS]
                if not unskipped:
                    print("\n  All remaining pages before cutoff, stopping.")
                    break

        page += 1
        time.sleep(0.3)

    print(f"\n{'='*50}")
    print(f"Final: {total_new} new, {total_dup} dup, {total_skip} skip, {total_fail} fail")

    if total_new > 0:
        print("\nRebuilding FTS index...")
        try:
            cur.execute(f"DELETE FROM gov_search WHERE site_name='{SITE_NAME}'")
            cur.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            print("  FTS index rebuilt.")
        except Exception as e:
            print(f"  [ERR] FTS rebuild: {e}")

    conn.close()
    print("Done.")


if __name__ == "__main__":
    main()
