#!/usr/bin/env python3
"""
crawl_lbx.py — 雷波县人民政府·环评公告
========================================
TRS CMS，createPageHTML(N) 分页，0-indexed
首页 index.html 静态内含文章列表

列表：<li><a href="./YYYYMM/tYYYYMMDD_id.html" title="TITLE">TITLE</a></li>
       日期从 URL 路径提取
详情：<td id="xilan_cont"> → <div class="TRS_UEDITOR"> 正文
       <meta name="PubDate" content="YYYY-MM-DD HH:MM" /> 日期

用法:
    python3 crawl_lbx.py             # 全量
    python3 crawl_lbx.py 1           # 增量（只爬首页）
"""

import os
import re
import sys
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

# ── DB ──────────────────────────────────────────────
DB_PATH = os.environ.get("SEARCH_DB", os.environ.get("GOV_DB_PATH", "/root/search.db"))

# ── URLs ────────────────────────────────────────────
BASE_URL = "http://www.lbx.gov.cn"
LIST_PATH = "/xxgk/zdxxgk/hjbh/hpgg/"
LIST_URL = BASE_URL + LIST_PATH
SITE_NAME = "雷波县人民政府"
SOURCE = "雷波县环评公告"
PAGES = 4  # createPageHTML(4,...)

# ── Headers ─────────────────────────────────────────
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
}

# ── helpers ─────────────────────────────────────────
def log(msg):
    print(f"[lbx] {msg}")

def extract_date_from_url(url):
    """Extract date from URL like ./202606/t20260612_2991748.html → 2026-06-12"""
    m = re.search(r'/t(\d{4})(\d{2})(\d{2})_', url)
    if m:
        return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
    return ""

def clean_title(title):
    """Clean title if needed"""
    return title.strip()

def build_list_url(page_index):
    """page_index: 0-based. 0 → index.html, 1 → index_1.html, etc."""
    if page_index == 0:
        return LIST_URL
    return f"{BASE_URL}{LIST_PATH}index_{page_index}.html"

def fetch_list_page(page_index):
    """Fetch one list page, return [(url, title, date_str), ...]"""
    url = build_list_url(page_index)
    log(f"Fetching list page {page_index}: {url}")
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR fetching list page: {e}")
        return []

    soup = BeautifulSoup(html, "html.parser")
    items = []

    # Find <li> with article links
    for li in soup.find_all("li", style=re.compile(r"WIDTH:\s*100%")):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        href = a["href"].strip()
        # Only local article links
        if not href.startswith("./") or "/t" not in href:
            continue

        title = a.get("title", "") or a.get_text(strip=True)
        full_url = urljoin(LIST_URL, href)
        date_str = extract_date_from_url(href)
        items.append((full_url, title, date_str))

    log(f"  Found {len(items)} articles")
    return items

def fetch_detail(url):
    """Return (content_html, date_str) or (None, None)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR fetching {url}: {e}")
        return None, None

    # Content: <td id="xilan_cont"> → <div class="TRS_UEDITOR">
    soup = BeautifulSoup(html, "html.parser")
    content_html = ""

    td = soup.find("td", id="xilan_cont")
    if td:
        # Find the TRS_UEDITOR div inside
        editor = td.find("div", class_=re.compile(r"TRS_UEDITOR"))
        if editor:
            content_html = str(editor)
        else:
            # Fallback: entire td content
            content_html = str(td)

    # Date: <meta name="PubDate" content="2026-06-12 11:12" />
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        raw = meta["content"].strip()
        # Format: "2026-06-12 11:12" or "2026-06-12"
        m = re.match(r"(\d{4}-\d{2}-\d{2})", raw)
        if m:
            date_str = m.group(1)

    # Fallback: from URL
    if not date_str:
        date_str = extract_date_from_url(url)

    if not content_html:
        log(f"  WARNING: no content for {url}")
        return None, date_str

    return content_html, date_str

def save_to_db(items):
    """Upsert records into search.db"""
    import sqlite3

    if not items:
        log("No items to save")
        return

    log(f"Opening DB: {DB_PATH}")
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    inserted = 0
    for url, title, date_str, content_html in items:
        try:
            c.execute(
                """INSERT OR REPLACE INTO gov_raw (site_name, page_url, title, publish_date, content, category, script_name) VALUES (?, ?, ?, ?, ?, ?, 'crawl_lbx.py')""",
                (
                    SITE_NAME,
                    url,
                    clean_title(title),
                    date_str,
                    content_html,
                    SOURCE,
                ),
            )
            inserted += 1
        except Exception as e:
            log(f"  DB error for {url}: {e}")

    conn.commit()
    conn.close()
    log(f"Saved {inserted}/{len(items)} records to DB")

def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] == "1"
    three_years_ago = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
    log(f"3-year cutoff: {three_years_ago}")

    # Collect articles from all pages (stop early if beyond 3 years)
    all_articles = []
    for pi in range(PAGES):
        articles = fetch_list_page(pi)
        if not articles:
            log(f"  No more articles, stopping at page {pi}")
            break

        # Filter 3-year
        filtered = [(u, t, d) for u, t, d in articles if d and d >= three_years_ago]
        all_articles.extend(filtered)

        # If incremental mode, only page 0
        if incremental:
            log("Incremental mode: page 0 only")
            break

        # Check if we went past 3-year mark
        if filtered and filtered[-1][2] < (datetime.now() - timedelta(days=365 * 2)).strftime("%Y-%m-%d"):
            # Remaining pages likely all before cutoff
            remaining_young = sum(1 for _, _, d in articles if d and d >= three_years_ago)
            if remaining_young == 0:
                log(f"  Page {pi}: all articles before cutoff, stopping")
                break

    log(f"Total articles after 3-year filter: {len(all_articles)}")

    # Fetch details
    results = []
    for i, (url, title, date_str) in enumerate(all_articles, 1):
        log(f"[{i}/{len(all_articles)}] {title[:50]}...")
        content_html, detail_date = fetch_detail(url)
        final_date = detail_date or date_str
        if content_html:
            results.append((url, title, final_date, content_html))
        else:
            log(f"  WARNING: saving {url} without content")
            results.append((url, title, final_date, ""))

    # Save
    save_to_db(results)

    # Summary
    log(f"\n{'='*50}")
    log(f"Done. Total: {len(results)} articles")
    if results:
        dates = sorted([r[2] for r in results if r[2]])
        log(f"Date range: {dates[0]} ~ {dates[-1]}")

    # Rebuild FTS (required for search)
    try:
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("""
            INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
            SELECT rowid, title, site_name, 
                   CASE WHEN length(content) > 200 THEN substr(content, 1, 200) ELSE content END
            FROM gov_raw WHERE category = ?
        """, (SOURCE,))
        conn.commit()
        conn.close()
        log("FTS index rebuilt for new entries")
    except Exception as e:
        log(f"FTS rebuild note: {e}")

    log(f"{'='*50}")


if __name__ == "__main__":
    main()
