#!/usr/bin/env python3
"""
crawl_jinan_epb_new.py — 济南市生态环境局·新栏目
=============================================
CMS: JCMS (大汉版通)，API 驱动

栏目清单：
  col32954 → 历城分局·拟审查项目公示
  col32967 → 商河分局·项目受理公示
  col10489 → 市局（本级）·项目受理公示

用法:
    python3 crawl_jinan_epb_new.py               # 全量（跑全部3个栏目）
    python3 crawl_jinan_epb_new.py 1             # 增量（仅最新页）
"""

import os
import re
import sys
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

BASE_URL = "https://jnepb.jinan.gov.cn"
SITE_NAME = "济南市生态环境局"

API_URL = f"{BASE_URL}/api-gateway/jpaas-publish-server/front/page/build/unit"
API_BASE = {
    "parseType": "bulidstatic",
    "webId": "21",
    "tplSetId": "pMB9pkaz19HYYbLAOTZtY",
    "pageType": "column",
    "tagId": "信息列表",
}

COLUMNS = [
    {
        "pageId": "32954",
        "column_name": "拟审查项目公示·历城分局",
        "referer": f"{BASE_URL}/col/col32954/index.html",
        "label": "[jn_lc_nsc]",
    },
    {
        "pageId": "32967",
        "column_name": "项目受理公示·商河分局",
        "referer": f"{BASE_URL}/col/col32967/index.html",
        "label": "[jn_sh_slgs]",
    },
    {
        "pageId": "10489",
        "column_name": "项目受理公示·市局",
        "referer": f"{BASE_URL}/col/col10489/index.html",
        "label": "[jn_sj_slgs]",
    },
]

HEADERS_TEMPLATE = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json",
}

MAX_WORKERS = 20
PAGE_SIZE = 50


def log(label, msg):
    print(f"{label} {msg}")


def fetch_articles_page(page_no, col):
    """Fetch one page of article list from API. Returns (items, total)."""
    param_json = json.dumps({"pageNo": page_no, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    params = dict(API_BASE)
    params["pageId"] = col["pageId"]
    params["paramJson"] = param_json

    headers = dict(HEADERS_TEMPLATE)
    headers["Referer"] = col["referer"]

    try:
        resp = requests.get(API_URL, headers=headers, params=params, timeout=30)
        data = resp.json()
        html = data.get("data", {}).get("html", "")
    except Exception as e:
        log(col["label"], f"ERROR API page {page_no}: {e}")
        return [], 0

    items = []
    total = 0

    m = re.search(r'count="(\d+)"', html)
    if m:
        total = int(m.group(1))

    for m in re.finditer(
        r'<a\s+href="([^"]+)"\s+title="([^"]+)"\s+target="_blank">',
        html
    ):
        href = m.group(1)
        title = m.group(2).strip()

        entry = html[m.start():m.end() + 200]
        date_m = re.search(r'\[(\d{4}-\d{2}-\d{2})\]', entry)
        date_str = date_m.group(1) if date_m else ""

        full_url = href if href.startswith("http") else urljoin(BASE_URL, href)
        items.append((full_url, title, date_str))

    return items, total


def fetch_detail(url, headers):
    """Return (content_html, date_str, title) or (None, None, None)."""
    try:
        resp = requests.get(url, headers=headers, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log("[detail]", f"ERROR {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # Title
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta_title["content"].strip() if meta_title and meta_title.get("content") else ""

    # Date
    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.match(r"(\d{4}-\d{2}-\d{2})", meta_date["content"].strip())
        if m:
            date_str = m.group(1)

    # Content
    content_div = soup.find(
        "div",
        class_=lambda c: c and "article" in (c if isinstance(c, str) else " ".join(c))
    )
    if not content_div:
        bt_article = soup.find("div", class_="bt-article-02")
        if bt_article:
            zoom = bt_article.find(id="zoom")
            content_div = zoom if zoom else bt_article
    content_html = str(content_div) if content_div else ""

    if not content_html:
        log("[detail]", f"WARNING: no content for {url}")
        return None, date_str, title

    return content_html, date_str, title


def fetch_detail_batch(urls, headers):
    results = {}
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
        fut_map = {ex.submit(fetch_detail, url, headers): url for url in urls}
        for fut in as_completed(fut_map):
            url = fut_map[fut]
            try:
                results[url] = fut.result()
            except Exception:
                results[url] = (None, None, None)
    return results


def save_to_db(items, column_name):
    """items: list of (url, title, date_str, content_html)"""
    if not items:
        return 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for url, title, date_str, content_html in items:
        try:
            c.execute(
                """INSERT OR REPLACE INTO gov_raw (site_name, page_url, title, publish_date, content, category, script_name) VALUES (?, ?, ?, ?, ?, ?, 'crawl_jinan_epb_new.py')""",
                (SITE_NAME, url, title, date_str, content_html, column_name),
            )
            inserted += 1
        except Exception as e:
            log("[db]", f"DB error for {url}: {e}")
    conn.commit()
    conn.close()
    return inserted


def rebuild_fts(column_name):
    try:
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute(
            """INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
               SELECT rowid, title, site_name,
                      CASE WHEN length(content) > 200 THEN substr(content, 1, 200) ELSE content END
               FROM gov_raw WHERE category = ?""",
            (column_name,),
        )
        conn.commit()
        conn.close()
        return True
    except Exception as e:
        log("[fts]", f"FTS rebuild note: {e}")
        return False


def crawl_column(col, incremental):
    label = col["label"]
    col_name = col["column_name"]
    three_years_ago = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")

    headers = dict(HEADERS_TEMPLATE)
    headers["Referer"] = col["referer"]

    log(label, f"=== {col_name} (pageId={col['pageId']}) ===")
    log(label, f"3-year cutoff: {three_years_ago}")

    # Phase 1: list pages
    all_articles = []
    total_expected = 0
    page = 1

    while True:
        items, total = fetch_articles_page(page, col)
        if page == 1:
            total_expected = total
            log(label, f"Expected total: {total_expected}")

        if not items:
            if page == 1:
                log(label, "No articles found")
                return 0
            break

        filtered = [(u, t, d) for u, t, d in items if d and d >= three_years_ago]
        all_articles.extend(filtered)

        if incremental:
            log(label, f"Incremental: page 1, {len(filtered)} articles")
            break

        if filtered and len(filtered) < len(items):
            log(label, f"Page {page}: some items before cutoff, stopping")
            break

        if page * PAGE_SIZE >= total_expected:
            break

        page += 1

    log(label, f"Articles to process: {len(all_articles)}")

    if not all_articles:
        return 0

    # Phase 2: details
    results = []
    batch_size = MAX_WORKERS * 3
    for batch_start in range(0, len(all_articles), batch_size):
        batch = all_articles[batch_start:batch_start + batch_size]
        batch_urls = [u for u, _, _ in batch]
        log(label, f"  Details [{batch_start+1}-{batch_start+len(batch)}/{len(all_articles)}]...")
        detail_map = fetch_detail_batch(batch_urls, headers)
        for url, title, date_str in batch:
            content_html, detail_date, detail_title = detail_map.get(url, (None, None, None))
            final_title = detail_title or title
            final_date = detail_date or date_str
            results.append((url, final_title, final_date, content_html or ""))

    # Save
    saved = save_to_db(results, col_name)
    log(label, f"Saved: {saved}/{len(results)}")

    if results:
        dates = sorted([r[2] for r in results if r[2]])
        log(label, f"Date range: {dates[0] if dates else 'N/A'} ~ {dates[-1] if dates else 'N/A'}")

    rebuild_fts(col_name)
    return saved


def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] == "1"
    total_saved = 0

    for col in COLUMNS:
        saved = crawl_column(col, incremental)
        total_saved += saved

    print(f"\n{'=' * 50}")
    print(f"[jinan_epb_new] All done! Total saved: {total_saved}")


if __name__ == "__main__":
    main()
