#!/usr/bin/env python3
"""葆华环保 - 环评公示 爬虫
Spring Boot + UEditor, POST /getNewsList API, 5条/页, 共64页320条
正文 content 字段直接在列表API返回中
"""

import os
import sys
import requests
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATE_ID = 17
PAGE_SIZE = 5
SITE_NAME = "baohuahb.cn-环评公示"
BASE_URL = "http://www.baohuahb.cn"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Content-Type": "application/x-www-form-urlencoded",
}

# 3年截止日期
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
print(f"[info] 日期过滤: >= {CUTOFF_DATE}")


def get_total_pages():
    """获取总页数"""
    try:
        r = requests.post(f"{BASE_URL}/getNewsList", data={
            "cate_id": CATE_ID,
            "currentPage": 1,
            "totalPage": PAGE_SIZE,
        }, headers=HEADERS, timeout=15)
        data = r.json()
        if data.get("code") == "0":
            total = data["data"]["totalPage"]
            print(f"[info] 总页数: {total}, 总条数: {data['data']['totalRow']}")
            return total
    except Exception as e:
        print(f"[error] 获取总页数失败: {e}")
    return 0


def fetch_page(page):
    """获取单页列表，返回条目列表"""
    try:
        r = requests.post(f"{BASE_URL}/getNewsList", data={
            "cate_id": CATE_ID,
            "currentPage": page,
            "totalPage": PAGE_SIZE,
        }, headers=HEADERS, timeout=15)
        data = r.json()
        if data.get("code") == "0":
            return data["data"].get("list", [])
    except Exception as e:
        print(f"[error] 第{page}页请求失败: {e}")
    return []


def main():
    import sqlite3

    total_pages = get_total_pages()
    if not total_pages:
        print("[error] 无法获取总页数，退出")
        sys.exit(1)

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    last_date = ""

    for page in range(1, total_pages + 1):
        items = fetch_page(page)
        if not items:
            print(f"[warn] 第{page}页无数据，跳过")
            continue

        page_new = 0
        page_skip = 0
        for item in items:
            pub_date = item.get("create_time", "")
            if pub_date < CUTOFF_DATE:
                page_skip += 1
                continue

            title = item.get("title", "").strip()
            content = item.get("content", "").strip()
            page_url = f"{BASE_URL}{item.get('detailUrl', '')}"
            last_date = pub_date

            try:
                c.execute("""
                    INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (page_url, title, content, SITE_NAME, pub_date, title, "环评公示"))
                if c.rowcount > 0:
                    page_new += 1
            except Exception as e:
                print(f"[error] 入库失败: {e}")

        conn.commit()
        new_count += page_new
        skip_count += page_skip
        print(f"[page {page}/{total_pages}] +{page_new} new, {page_skip} skipped, cumulative: {new_count} new")

    conn.close()
    print(f"\n[完成] 葆华环保: 新增 {new_count} 条, 跳过 {skip_count} 条(超3年)")
    print(f"[提示] 最近日期: {last_date}")


if __name__ == "__main__":
    main()
