#!/usr/bin/env python3
"""
南充市顺庆区人民政府 — 调查征集（意见征集）
Custom IGI/opinion API (JSON)
https://www.shunqing.gov.cn/hdjl/opinion/
"""
import json
import re
import os
import urllib3
import requests
import sqlite3
from datetime import datetime, timezone

# Disable SSL warnings (self-signed cert on shunqing.gov.cn)
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "南充市顺庆区人民政府"
BASE_URL = "https://www.shunqing.gov.cn"
API_URL = BASE_URL + "/IGI/opinion/web/list"
CUTOFF_DATE = datetime(2023, 6, 18, tzinfo=timezone.utc)

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": BASE_URL + "/hdjl/opinion/",
}

PAGE_SIZE = 10
SITE_ID = 47


def api_get(page_index, timeout=15):
    """Call the opinion list API."""
    params = {
        "siteId": SITE_ID,
        "pageIndex": page_index,
        "pageSize": PAGE_SIZE,
        "orderBy": "startTime_desc",
    }
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=timeout, verify=False)
        r.raise_for_status()
        data = r.json()
        if data.get("statusCode") == 200:
            return data.get("datas", {}).get("data", []), data.get("datas", {}).get("pageInfo", {})
        return [], {}
    except Exception as e:
        print(f"  API error (page {page_index}): {e}")
        return None, {}


def parse_timestamp(ts_str):
    """Parse timestamp string (milliseconds) to date string."""
    try:
        ts = int(float(ts_str))
        if ts > 1_000_000_000_000:  # milliseconds
            ts = ts // 1000
        return datetime.fromtimestamp(ts, tz=timezone.utc)
    except:
        return None


def insert_db(items):
    """Insert items into search.db."""
    if not items:
        print("  无数据插入")
        return 0
    conn = sqlite3.connect(DB_PATH, timeout=10)
    c = conn.cursor()
    inserted = 0
    for item in items:
        try:
            clean_text = re.sub(r'<[^>]+>', '', item["content"]).strip()
            c.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (title, page_url, site_name, summary, content, publish_date)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (item["title"], item["url"], SITE_NAME, clean_text[:500],
                 item["content"], item["date"]),
            )
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB insert error: {e}")
    conn.commit()
    conn.close()
    return inserted


def rebuild_fts():
    """Rebuild FTS index."""
    for attempt in range(3):
        try:
            conn = sqlite3.connect(DB_PATH, timeout=10)
            c = conn.cursor()
            c.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            conn.close()
            print("  FTS 重建完成")
            return
        except Exception as e:
            if attempt < 2:
                import time
                time.sleep(2)
            print(f"  FTS 重建第{attempt+1}次失败: {e}")


def main():
    print(f"=== 开始爬取 {SITE_NAME} - 调查征集 ===")
    print(f"截止日期: {CUTOFF_DATE.strftime('%Y-%m-%d')}")

    # Step 1: Get first page for total count
    items, page_info = api_get(1)
    if items is None:
        print("  ❌ 无法获取第一页数据")
        return

    total_results = int(page_info.get("totalResults", 0))
    total_pages = int(page_info.get("totalPages", 0))
    print(f"  总数: {total_results}, 页数: {total_pages}")

    # Step 2: Get all pages
    all_items = list(items)
    for page_no in range(2, total_pages + 1):
        items, _ = api_get(page_no)
        if items:
            all_items.extend(items)
        if page_no % 3 == 0:
            print(f"  已扫描 {page_no}/{total_pages} 页, 累计 {len(all_items)} 条")

    print(f"  列表扫描完成: {len(all_items)} 条")

    # Step 3: Filter by date (use beginTime as publish date)
    from datetime import timedelta
    cutoff_ts = CUTOFF_DATE.timestamp()

    filtered = []
    for item in all_items:
        bt = parse_timestamp(item.get("beginTime", "0"))
        if bt and bt.timestamp() >= cutoff_ts:
            date_str = bt.strftime("%Y-%m-%d")
            content = item.get("htmlContent", "")
            if not content.strip():
                print(f"  跳过空内容: {item.get('theme','')[:40]}")
                continue
            filtered.append({
                "title": item.get("theme", ""),
                "url": item.get("publishUrl", ""),
                "date": date_str,
                "content": content,
            })

    print(f"  3年内: {len(filtered)} 条")

    # Debug: check first item
    if filtered:
        print(f"  示例: {filtered[0]['title'][:40]} -> {filtered[0]['url']} (content: {len(filtered[0]['content'])} chars)")

    # Step 4: Insert into DB (htmlContent is already in the API!)
    inserted = insert_db(filtered)
    print(f"  入库: {inserted} 条")

    # Step 5: Rebuild FTS
    rebuild_fts()

    print(f"=== 完成: 共 {len(filtered)} 条成功, 新增 {inserted} 条 ===")


if __name__ == "__main__":
    main()
