#!/usr/bin/env python3
"""
crawl_wjq.py — 第六师五家渠市人民政府·通知公告
=============================================
JPaaS CMS，API JSON 分页

列表 API：/api-gateway/jpaas-publish-server/front/page/build/unit
  paramJson={"pageNo":N,"pageSize":12} 控制分页
  返回 JSON { data: { html: "..." } }

详情：<div class="zw_content"> 正文
      <meta name="PubDate" content="2026-06-12 19:55" /> 日期

用法:
    python3 crawl_wjq.py             # 全量
    python3 crawl_wjq.py 1           # 增量（只爬首页）
"""

import os
import re
import sys
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

# ── DB ──────────────────────────────────────────────
DB_PATH = os.environ.get("SEARCH_DB", os.environ.get("GOV_DB_PATH", "/root/search.db"))

# ── URLs ────────────────────────────────────────────
BASE_URL = "https://www.wjq.gov.cn"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_NAME = "第六师五家渠市人民政府"
SOURCE = "五家渠市通知公告"

# API static params
API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "gXiRWWCDbJQMHYm3xJLgF",
    "tplSetId": "GLcGu6Emt0FbNhQGT3B8e",
    "pageType": "column",
    "tagId": "当前栏目的列表",
    "editType": "null",
    "pageId": "Wb3cNQs5RoDi7h7NnZA55",
}
PAGE_SIZE = 12
MAX_PAGES = 160  # ~1845 records / 12

# ── Headers ─────────────────────────────────────────
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
}

# ── helpers ─────────────────────────────────────────
def log(msg):
    print(f"[wjq] {msg}")

def clean_title(title):
    return title.strip()

def fetch_list_page(page):
    """Fetch one page via API, return [(url, title, date_str), ...]"""
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": PAGE_SIZE},
                                     ensure_ascii=False)

    try:
        resp = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
        data = resp.json()
    except Exception as e:
        log(f"  ERROR fetching page {page}: {e}")
        return [], False

    if not data.get("success") or not data.get("data", {}).get("html"):
        return [], False

    html = data["data"]["html"]
    soup = BeautifulSoup(html, "html.parser")
    items = []

    for li in soup.find_all("li"):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        href = a["href"].strip()
        if not href.startswith("/"):
            href = "/" + href.lstrip("./")

        # Title from <p title="..."> or <p> text
        p = a.find("p")
        title = ""
        if p:
            title = p.get("title", "") or p.get_text(strip=True)

        # Date from <span>
        span = a.find("span")
        date_str = span.get_text(strip=True) if span else ""

        full_url = urljoin(BASE_URL, href)
        items.append((full_url, clean_title(title), date_str))

    return items, len(items) > 0

def fetch_detail(url):
    """Return (content_html, date_str) or (None, None)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR fetching {url}: {e}")
        return None, None

    soup = BeautifulSoup(html, "html.parser")

    # Content: <div class="zw_content">
    content_html = ""
    div = soup.find("div", class_="zw_content")
    if div:
        content_html = str(div)

    # Date: <meta name="PubDate" content="2026-06-12 19:55" />
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        raw = meta["content"].strip()
        m = re.match(r"(\d{4}-\d{2}-\d{2})", raw)
        if m:
            date_str = m.group(1)

    if not content_html:
        log(f"  WARNING: no content for {url}")
        return None, date_str

    return content_html, date_str

def save_to_db(items):
    """Upsert records into search.db"""
    import sqlite3

    if not items:
        log("No items to save")
        return 0

    log(f"Opening DB: {DB_PATH}")
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    inserted = 0
    for url, title, date_str, content_html in items:
        try:
            c.execute(
                """INSERT OR REPLACE INTO gov_raw (site_name, page_url, title, publish_date, content, category, script_name) VALUES (?, ?, ?, ?, ?, ?, 'crawl_wjq.py')""",
                (
                    SITE_NAME,
                    url,
                    title,
                    date_str,
                    content_html,
                    SOURCE,
                ),
            )
            inserted += 1
        except Exception as e:
            log(f"  DB error for {url}: {e}")

    conn.commit()
    conn.close()
    log(f"Saved {inserted}/{len(items)} records to DB")
    return inserted

def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] == "1"
    three_years_ago = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
    log(f"3-year cutoff: {three_years_ago}")

    # Phase 1: Collect article URLs from API pages
    all_articles = []
    for page in range(1, MAX_PAGES + 1):
        articles, has_data = fetch_list_page(page)
        if not has_data:
            log(f"Page {page}: no data, stopping")
            break

        # Filter 3-year
        filtered = [(u, t, d) for u, t, d in articles if d and d >= three_years_ago]
        all_articles.extend(filtered)

        if incremental:
            log(f"Incremental: page 1 only, {len(filtered)} articles")
            break

        # Stop early if all items on this page are before cutoff
        if filtered and len(filtered) < len(articles):
            # Some items were filtered - check if ALL are before cutoff
            earliest = min(d for _, _, d in articles if d)
            if earliest < three_years_ago:
                # This page straddles the cutoff; next page will be all before
                newest = max(d for _, _, d in articles if d)
                if newest < three_years_ago:
                    break

        if page % 30 == 0:
            log(f"  Scanned page {page}, total {len(all_articles)} articles")

    log(f"Total articles: {len(all_articles)}")

    # Phase 2: Fetch details
    results = []
    for i, (url, title, date_str) in enumerate(all_articles, 1):
        log(f"[{i}/{len(all_articles)}] {title[:50]}...")
        content_html, detail_date = fetch_detail(url)
        final_date = detail_date or date_str
        if content_html:
            results.append((url, title, final_date, content_html))
        else:
            results.append((url, title, final_date, ""))

    # Phase 3: Save
    saved = save_to_db(results)

    # Summary
    log(f"\n{'='*50}")
    log(f"Done. Total: {len(results)} articles (saved: {saved})")
    if results:
        dates = sorted([r[2] for r in results if r[2]])
        log(f"Date range: {dates[0]} ~ {dates[-1]}")

    # Rebuild FTS
    try:
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("""
            INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
            SELECT rowid, title, site_name, 
                   CASE WHEN length(content) > 200 THEN substr(content, 1, 200) ELSE content END
            FROM gov_raw WHERE category = ?
        """, (SOURCE,))
        conn.commit()
        conn.close()
        log("FTS index rebuilt")
    except Exception as e:
        log(f"FTS rebuild note: {e}")

    log(f"{'='*50}")


if __name__ == "__main__":
    main()
