#!/usr/bin/env python3
"""江油市 - 公示公告 爬虫
同CMS，API: /common/search/590da82f..., 862条共44页
详情页 <div class="article-content" id="zoomcon"> 正文
"""

import os
import sys
import re
import requests
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "jiangyou.gov.cn-公示公告"
BASE_URL = "https://www.jiangyou.gov.cn"
CHANNEL_ID = "590da82f40934cf2afa46bdc7c87aae0"
API_URL = f"{BASE_URL}/common/search/{CHANNEL_ID}"
PAGE_SIZE = 20

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
print(f"[info] 日期过滤: >= {CUTOFF_DATE}")


def fetch_api_page(page):
    """调用AJAX API获取列表"""
    try:
        r = requests.get(f"{API_URL}?_isJson=true&_pageSize={PAGE_SIZE}&_template=index&page={page}",
                         headers=HEADERS, timeout=15)
        return r.json()
    except Exception as e:
        print(f"[error] API第{page}页请求失败: {e}")
        return None


def fetch_detail_content(url):
    """从详情页提取正文"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        html = r.text
        m = re.search(r'<div class="article-content" id="zoomcon">([\s\S]*?)</div>\s*<div class="', html)
        if m:
            return m.group(1).strip()
        return ""
    except Exception as e:
        print(f"  [warn] 详情页失败: {url[-50:]}, {e}")
        return ""


def main():
    import sqlite3

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute("PRAGMA journal_mode=WAL")

    data = fetch_api_page(1)
    if not data:
        print("[error] 无法获取数据，退出")
        sys.exit(1)

    total = data["data"]["total"]
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f"[info] 总条数: {total}, 总页数: {total_pages}")

    new_count = 0
    old_count = 0
    dup_count = 0
    err_count = 0

    for page in range(1, total_pages + 1):
        data = fetch_api_page(page)
        if not data:
            continue

        results = data["data"]["results"]
        page_new = 0
        for item in results:
            title = item.get("title", "").strip()
            rel_url = item.get("url", "")
            date_str = item.get("publishedTimeStr", "")
            if not date_str:
                continue
            date = date_str[:10]
            if date < CUTOFF_DATE:
                old_count += 1
                continue

            url = BASE_URL + rel_url if rel_url.startswith("/") else rel_url

            c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone()[0] > 0:
                dup_count += 1
                continue

            content = fetch_detail_content(url)
            if not content:
                content = f"[无正文] {title}"

            try:
                c.execute("""
                    INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (url, title, content, SITE_NAME, date, title, "公示公告"))
                if c.rowcount > 0:
                    page_new += 1
            except Exception as e:
                print(f"  [error] 入库失败: {e}")
                err_count += 1

        conn.commit()
        new_count += page_new
        print(f"[page {page}/{total_pages}] +{page_new} new, cumulative: {new_count} new")

    conn.close()
    print(f"\n[完成] 江油公示公告: 新增 {new_count} 条, 旧跳过 {old_count}, 重复 {dup_count}, 错误 {err_count}")


if __name__ == "__main__":
    main()
