#!/usr/bin/env python3
"""
crawl_chengwu.py — 成武县人民政府·环境影响评价公众参与
======================================================
CMS: ELS 服务系统，API 驱动

列表：POST /els-service/article/{pageNo}/{pageSize}
      JSON body: {"catas": ["1535192418234798080"]}
分页：10条/页
详情：/10086/{dwid}/{xxid}.html
      标题 <meta name="ArticleTitle">
      日期 <meta name="PubDate">
      正文 embedded in <script>var memo = "..."</script>

用法:
    python3 crawl_chengwu.py               # 全量
    python3 crawl_chengwu.py 1             # 增量
"""

import os
import re
import sys
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "成武县人民政府"
COLUMN_NAME = "环境影响评价公众参与"
BASE_URL = "http://www.chengwu.gov.cn"
API_URL = f"{BASE_URL}/els-service/article"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
    "Content-Type": "application/json;charset=utf-8",
}

HEADERS_HTML = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
}

MAX_WORKERS = 20
PAGE_SIZE = 10


def log(msg):
    print(f"[chengwu] {msg}")


def fetch_list_page(page_no):
    """Fetch one list page from API. Returns list of (xxid, dwid, title, date_str)."""
    url = f"{API_URL}/{page_no}/{PAGE_SIZE}"
    payload = {"catas": ["1535192418234798080"]}

    try:
        resp = requests.post(url, headers=HEADERS, json=payload, timeout=30)
        data = resp.json()
        contents = data.get("data", {}).get("contents", [])
    except Exception as e:
        log(f"  ERROR list page {page_no}: {e}")
        return []

    items = []
    for c in contents:
        xxid = c.get("xxid", "")
        dwid = c.get("dwid", "")
        title = c.get("subject", "").strip()
        date_str = c.get("fwdate", "")
        items.append((xxid, dwid, title, date_str))

    return items


def extract_memo_content(html):
    """Extract and decode var memo = '...' from script tag."""
    idx = html.find("var memo =")
    if idx < 0:
        return ""

    chunk = html[idx:]
    quote_start = chunk.index('"') + 1

    end = quote_start
    while end < len(chunk):
        if chunk[end] == '\\':
            end += 2
        elif chunk[end] == '"':
            rest = chunk[end+1:end+5].strip()
            if not rest or rest.startswith(';') or rest.startswith('\n') or rest.startswith('\r'):
                break
            end += 1
        else:
            end += 1

    memo_raw = chunk[quote_start:end]

    # Decode JS escapes
    try:
        memo_decoded = memo_raw.encode().decode("unicode_escape")
    except Exception:
        memo_decoded = memo_raw

    # Clean up: remove <\\/ to </, etc.
    memo_decoded = memo_decoded.replace("\\/", "/")
    memo_decoded = memo_decoded.replace("\\n", "\n")
    memo_decoded = memo_decoded.replace("\\t", "\t")

    return memo_decoded


def fetch_detail(xxid, dwid):
    """Return (content_html, date_str, title) or (None, None, None)."""
    url = f"{BASE_URL}/10086/{dwid}/{xxid}.html"

    try:
        resp = requests.get(url, headers=HEADERS_HTML, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # Title
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta_title["content"].strip() if meta_title and meta_title.get("content") else ""

    # Date
    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.match(r"(\d{4}-\d{2}-\d{2})", meta_date["content"].strip())
        if m:
            date_str = m.group(1)

    # Content from memo var
    content_html = extract_memo_content(html)

    if not content_html:
        log(f"  WARNING: no content for {url}")
        return None, date_str, title

    return content_html, date_str, title


def fetch_detail_batch(items):
    """items: list of (xxid, dwid, title, date_str)"""
    results = {}
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
        fut_map = {
            ex.submit(fetch_detail, xxid, dwid): (xxid, dwid)
            for xxid, dwid, _, _ in items
        }
        for fut in as_completed(fut_map):
            xxid, dwid = fut_map[fut]
            try:
                results[(xxid, dwid)] = fut.result()
            except Exception:
                results[(xxid, dwid)] = (None, None, None)
    return results


def save_to_db(rows):
    if not rows:
        return 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for url, title, date_str, content_html in rows:
        try:
            c.execute(
                """INSERT OR REPLACE INTO gov_raw
                   (site_name, page_url, title, publish_date, content, category)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (SITE_NAME, url, title, date_str, content_html, COLUMN_NAME),
            )
            inserted += 1
        except Exception as e:
            log(f"  DB error for {url}: {e}")
    conn.commit()
    conn.close()
    return inserted


def rebuild_fts():
    try:
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute(
            """INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
               SELECT rowid, title, site_name,
                      CASE WHEN length(content) > 200 THEN substr(content, 1, 200) ELSE content END
               FROM gov_raw WHERE category = ?""",
            (COLUMN_NAME,),
        )
        conn.commit()
        conn.close()
    except Exception as e:
        log(f"FTS note: {e}")


def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] == "1"
    three_years_ago = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
    log(f"3-year cutoff: {three_years_ago}")

    # Phase 1: collect from list API
    all_items = []
    page = 1

    while True:
        items = fetch_list_page(page)
        if not items:
            if page == 1:
                log("No articles found")
                return
            break

        log(f"Page {page}: {len(items)} articles")

        filtered = [(xxid, dwid, title, d) for xxid, dwid, title, d in items if d and d >= three_years_ago]
        all_items.extend(filtered)

        if incremental:
            log("Incremental: page 1 only")
            break

        if filtered and len(filtered) < len(items):
            log(f"Page {page}: some articles before cutoff, stopping")
            break

        page += 1

    log(f"Total articles to process: {len(all_items)}")

    if not all_items:
        log("No articles within 3 years")
        return

    # Phase 2: fetch details
    results = []
    batch_size = MAX_WORKERS * 3
    for batch_start in range(0, len(all_items), batch_size):
        batch = all_items[batch_start:batch_start + batch_size]
        log(f"  Details [{batch_start+1}-{batch_start+len(batch)}/{len(all_items)}]...")
        detail_map = fetch_detail_batch(batch)

        for xxid, dwid, title, date_str in batch:
            content_html, detail_date, detail_title = detail_map.get((xxid, dwid), (None, None, None))
            final_title = detail_title or title
            final_date = detail_date or date_str
            url = f"{BASE_URL}/10086/{dwid}/{xxid}.html"
            results.append((url, final_title, final_date, content_html or ""))

    # Save
    saved = save_to_db(results)
    log(f"Saved: {saved}/{len(results)}")

    if results:
        dates = sorted([r[2] for r in results if r[2]])
        log(f"Date range: {dates[0] if dates else 'N/A'} ~ {dates[-1] if dates else 'N/A'}")

    rebuild_fts()
    log("Done!")


if __name__ == "__main__":
    main()
