#!/usr/bin/env python3
import os
"""
寒亭区人民政府 - 公告公示 爬虫
http://www.hanting.gov.cn/htlist/?gggs
API: POST /els-service/article/{page}/15
"""
import re
import sys
import time
import json
import sqlite3
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin

BASE_URL = "http://www.hanting.gov.cn/htlist/?gggs"
DOMAIN = "www.hanting.gov.cn"
SITE_NAME = "寒亭区公告公示"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATAID = "1760120772858351616"  # gggs channel
MAX_PAGES = 80
INCREMENTAL = "--incremental" in sys.argv
YEAR_LIMIT = 3

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
           "Content-Type": "application/json;charset=utf-8"}

conn = sqlite3.connect(DB_PATH, timeout=60)
c = conn.cursor()

cutoff_date = datetime.now(timezone(timedelta(hours=8))).replace(tzinfo=None) - timedelta(days=365 * YEAR_LIMIT)

def fetch_list_api(page):
    url = f"http://www.hanting.gov.cn/els-service/article/{page}/15"
    try:
        r = requests.post(url, json={"dq":"98","order":"fwdate","catas":[CATAID]}, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        data = r.json()
        return data.get("data", {}).get("contents", []), data.get("data", {}).get("elementsTotal", 0)
    except Exception as e:
        print(f"  [API ERROR] page {page}: {e}")
        return None, 0

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  [ERROR] fetch {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # Title
    title = ""
    doctitle = soup.find("div", class_="doctitle")
    if doctitle:
        h1 = doctitle.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True)

    # Date
    date_str = ""
    docdate = soup.find("div", class_="docdate")
    if docdate:
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', docdate.get_text())
        if dm:
            date_str = dm.group(1)

    # Content from doccontent > ozoom
    content_html = ""
    doccontent = soup.find("div", class_="doccontent", id="doccontent")
    if doccontent:
        ozoom = doccontent.find("div", id="ozoom")
        if ozoom:
            content_html = str(ozoom)
        else:
            content_html = str(doccontent)
    
    if not content_html or content_html == "<div id=\"ozoom\"></div>":
        content_html = "<p>内容加载失败</p>"

    return title, content_html, date_str

def article_exists(page_url):
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url = ?", (page_url,))
    return c.fetchone()[0] > 0

def sync_fts(row_id, title, site_name):
    try:
        c.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, site_name, ""))
    except:
        pass

def insert_article(title, page_url, content_html, date_str):
    display_url = page_url.replace("http://", "").replace("https://", "")
    summary = ""

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None

    try:
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_hanting.py')""",
            (page_url, title, content_html, date_str, SITE_NAME, display_url, summary, ""))

        row_id = existing_id or c.lastrowid
        if row_id:
            sync_fts(row_id, title, SITE_NAME)
        return True
    except Exception as e:
        print(f"  [DB ERROR] {e}")
        return False

def main():
    total_new = 0
    total_updated = 0
    total_skipped = 0

    # Get total pages
    _, total = fetch_list_api(1)
    total_pages = min((total + 14) // 15, MAX_PAGES)
    print(f"  Total items: {total}, pages to crawl: 1-{total_pages}")

    for page in range(1, total_pages + 1):
        print(f"  List page {page}/{total_pages}")

        results, _ = fetch_list_api(page)
        if results is None:
            continue
        if not results:
            print(f"    No items")
            continue

        print(f"    Found {len(results)} items")

        for item in results:
            title = item.get("subject", "")
            api_date = item.get("fwdate", "")[:10] if item.get("fwdate") else ""

            # Build URL
            xxid = item.get("xxid", "")
            dwid = item.get("dwid", "")
            dq = item.get("dq", "98")
            api_url = item.get("url", "")
            
            if api_url:
                page_url = urljoin("http://www.hanting.gov.cn/", api_url)
            else:
                page_url = f"http://www.hanting.gov.cn/{dq}/{dwid}/{xxid}.html"

            if api_date:
                try:
                    item_date = datetime.strptime(api_date, "%Y-%m-%d")
                    if item_date < cutoff_date.replace(tzinfo=None):
                        total_skipped += 1
                        continue
                except:
                    pass

            exists = article_exists(page_url)
            if INCREMENTAL and exists:
                total_skipped += 1
                continue

            print(f"    Fetching: {title[:35]}...", end=" ")
            d_title, content_html, detail_date = fetch_detail(page_url)

            if not d_title:
                d_title = title
            if not detail_date and api_date:
                detail_date = api_date

            if insert_article(d_title, page_url, content_html, detail_date):
                if exists:
                    total_updated += 1
                    print("updated")
                else:
                    total_new += 1
                    print("inserted")
            else:
                print("FAILED")

            time.sleep(0.3)

        conn.commit()
        print(f"  [COMMIT] page {page} done")

    conn.close()
    print(f"\n  ✅ 完成: 新增 {total_new}, 更新 {total_updated}, 跳过 {total_skipped}")

if __name__ == "__main__":
    main()
