#!/usr/bin/env python3
"""
务川仡佬族苗族自治县人民政府 - 网上征集（环评公示）
TRS IGI 系统，API: /IGI/opinion/web/list
"""
import json
import time
import random
import datetime
import urllib.request
import urllib.error
import sqlite3
import os

DB_PATH = os.environ.get("SEARCH_DB", "/root/search.db")
SITE_NAME = "务川县人民政府-网上征集"
API_URL = "https://www.gzwuchuan.gov.cn/IGI/opinion/web/list"
PAGE_SIZE = 20
SITE_ID = 500273
CUTOFF_DATE = datetime.datetime(2023, 6, 17)  # 3年截点


def fetch_page(page):
    """获取一页列表数据，返回items列表"""
    params = f"pageIndex={page}&pageSize={PAGE_SIZE}&siteId={SITE_ID}&orderby=endTime_desc"
    url = f"{API_URL}?{params}"
    req = urllib.request.Request(
        url,
        headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
        },
    )
    try:
        with urllib.request.urlopen(req, timeout=15) as resp:
            data = json.loads(resp.read().decode("utf-8"))
            return data.get("datas", {}).get("data", [])
    except Exception as e:
        print(f"  [ERROR] 第{page}页请求失败: {e}")
        return []


def clean_html(html):
    """清理HTML，保留结构化标签"""
    if not html:
        return ""
    # 移除 <script> 和 <style>
    import re
    html = re.sub(r'<script[^>]*>.*?</script>', '', html, flags=re.DOTALL | re.IGNORECASE)
    html = re.sub(r'<style[^>]*>.*?</style>', '', html, flags=re.DOTALL | re.IGNORECASE)
    # 保留基本HTML结构标签
    return html.strip()


def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    page = 1
    total_inserted = 0
    total_skipped = 0
    reached_cutoff = False

    while not reached_cutoff:
        print(f"--- 第{page}页 ---")
        items = fetch_page(page)
        if not items:
            print(f"  无数据，结束")
            break

        for item in items:
            # 时间解析
            begin_ts = int(item.get("beginTime", 0)) / 1000
            pub_date = datetime.datetime.fromtimestamp(begin_ts)
            date_str = pub_date.strftime("%Y-%m-%d")

            # 3年过滤
            if pub_date < CUTOFF_DATE:
                print(f"  [跳过] {date_str} | {item['theme'][:40]}...（超3年）")
                total_skipped += 1
                reached_cutoff = True
                continue

            # 标题
            title = item.get("theme", "").strip()
            if not title:
                continue

            # URL
            page_url = item.get("publishUrl", "") or item.get("exLink", "")
            if not page_url:
                continue

            # 正文
            content = clean_html(item.get("htmlContent", ""))

            # source_url = 原始API(列表页)
            source_url = f"{API_URL}?pageIndex={page}&pageSize={PAGE_SIZE}&siteId={SITE_ID}&orderby=endTime_desc"

            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content) VALUES (?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, title, page_url, date_str, source_url, content),
                )
                if c.rowcount > 0:
                    total_inserted += 1
                    print(f"  [插入] {date_str} | {title[:50]}")
                else:
                    total_skipped += 1
                    print(f"  [重复] {date_str} | {title[:50]}")
            except Exception as e:
                print(f"  [DB ERROR] {e}")

        if len(items) < PAGE_SIZE:
            print(f"  最后一页（不足{PAGE_SIZE}条），结束")
            break

        page += 1
        time.sleep(random.uniform(0.3, 0.5))

    conn.commit()
    conn.close()

    print(f"\n===== 完成 =====")
    print(f"站点: {SITE_NAME}")
    print(f"新增: {total_inserted} 条")
    print(f"跳过(重复/超3年): {total_skipped} 条")


if __name__ == "__main__":
    main()
