#!/usr/bin/env python3
"""
crawl_guanxian.py — 冠县人民政府·通知公告
==========================================
聚合平台，汇集各部门通知公告
?page=N 分页，13条/页，~450页 / ~5900条（3年）

列表：<a href="/site_xxx/channel_x_xxx_6351/doc_xxx.html">
        <div>...<em>YYYY-MM-DD</em>...Title</div></a>
详情：<div class="main ovh"> 正文 | <meta name="PubDate"> 日期

用法:
    python3 crawl_guanxian.py             # 全量
    python3 crawl_guanxian.py 1           # 增量（只爬首页）
"""

import os
import re
import sys
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed

# ── DB ──────────────────────────────────────────────
DB_PATH = os.environ.get("SEARCH_DB", os.environ.get("GOV_DB_PATH", "/root/search.db"))

# ── URLs ────────────────────────────────────────────
BASE_URL = "http://www.guanxian.gov.cn"
LIST_URL = BASE_URL + "/channel_x_0.0_6351/"
SITE_NAME = "冠县人民政府"
SOURCE = "冠县通知公告"

# ── Headers ─────────────────────────────────────────
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
}

MAX_WORKERS = 20  # parallel detail fetch
MAX_PAGES = 500   # safety limit

# ── helpers ─────────────────────────────────────────
def log(msg):
    print(f"[guanxian] {msg}")

def clean_title(title):
    return title.strip()

def fetch_list_page(page):
    """Return [(url, title, date_str), ...]"""
    url = f"{LIST_URL}?page={page}" if page > 1 else LIST_URL
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR page {page}: {e}")
        return [], False

    items = []
    # Find <a> tags with doc_ pattern
    for m in re.finditer(
        r'<a\s+href=\"([^\"]*doc_[^\"]+\.html)\"[^>]*>.*?</a>',
        html, re.DOTALL
    ):
        href = m.group(1)
        a_html = m.group(0)
        text = re.sub(r'<[^>]+>', '', a_html).strip()
        text = re.sub(r'\s+', ' ', text).strip()

        # Extract date
        date_str = ""
        d = re.search(r'(\d{4}-\d{2}-\d{2})', text)
        if d:
            date_str = d.group(1)

        # Extract title (remove date, remove source department after title)
        title = re.sub(r'\d{4}-\d{2}-\d{2}', '', text).strip()
        title = clean_title(title)

        # Build absolute URL
        if href.startswith("http"):
            full_url = href
        else:
            full_url = urljoin(BASE_URL, href)

        items.append((full_url, title, date_str))

    return items, len(items) > 0

def fetch_detail(url):
    """Return (content_html, date_str) or (None, None)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR {url}: {e}")
        return None, None

    soup = BeautifulSoup(html, "html.parser")

    # Content: <div class="main ovh">
    content_html = ""
    div = soup.find("div", class_=lambda c: c and "main" in c and "ovh" in c)
    if not div:
        div = soup.find("div", class_="main")
    if div:
        content_html = str(div)

    # Date: <meta name="PubDate" content="2026-06-12 10:02:26">
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.match(r"(\d{4}-\d{2}-\d{2})", meta["content"].strip())
        if m:
            date_str = m.group(1)

    if not content_html:
        log(f"  WARNING: no content for {url}")
        return None, date_str

    return content_html, date_str

def fetch_detail_batch(urls):
    """Fetch multiple detail pages in parallel"""
    results = {}
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
        fut_map = {ex.submit(fetch_detail, url): url for url in urls}
        for fut in as_completed(fut_map):
            url = fut_map[fut]
            try:
                results[url] = fut.result()
            except Exception as e:
                log(f"  ERROR in thread for {url}: {e}")
                results[url] = (None, None)
    return results

def save_to_db(items):
    import sqlite3
    if not items:
        return 0
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for url, title, date_str, content_html in items:
        try:
            c.execute(
                """INSERT OR REPLACE INTO gov_raw (site_name, page_url, title, publish_date, content, category, script_name) VALUES (?, ?, ?, ?, ?, ?, 'crawl_guanxian.py')""",
                (SITE_NAME, url, title, date_str, content_html, SOURCE),
            )
            inserted += 1
        except Exception as e:
            log(f"  DB error for {url}: {e}")
    conn.commit()
    conn.close()
    return inserted

def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] == "1"
    three_years_ago = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
    log(f"3-year cutoff: {three_years_ago}")

    # Phase 1: Scan list pages
    all_articles = []
    for page in range(1, MAX_PAGES + 1):
        articles, has_data = fetch_list_page(page)
        if not has_data:
            log(f"Page {page}: no data, stopping")
            break

        # Filter 3-year
        filtered = [(u, t, d) for u, t, d in articles if d and d >= three_years_ago]
        all_articles.extend(filtered)

        if incremental:
            log(f"Incremental: page 1, {len(filtered)} articles")
            break

        # Stop when all past cutoff
        if articles and not filtered:
            oldest_on_page = max([d for _, _, d in articles if d] or [""])
            if oldest_on_page and oldest_on_page < three_years_ago:
                log(f"Page {page}: all before cutoff, stopping")
                break

        if page % 50 == 0:
            log(f"  Page {page}: {len(all_articles)} articles so far")

    log(f"Total articles: {len(all_articles)}")

    # Phase 2: Fetch details (parallel batches)
    results = []
    batch_size = MAX_WORKERS * 3  # 60 per batch for smooth progress
    for batch_start in range(0, len(all_articles), batch_size):
        batch = all_articles[batch_start:batch_start + batch_size]
        batch_urls = [u for u, _, _ in batch]
        log(f"  Fetching details [{batch_start+1}-{batch_start+len(batch)}/{len(all_articles)}]...")

        detail_map = fetch_detail_batch(batch_urls)

        for url, title, date_str in batch:
            content_html, detail_date = detail_map.get(url, (None, None))
            final_date = detail_date or date_str
            results.append((url, title, final_date, content_html or ""))

    # Phase 3: Save
    saved = save_to_db(results)

    # Summary
    log(f"\n{'='*50}")
    log(f"Done. Total: {len(results)} articles (saved: {saved})")
    if results:
        dates = sorted([r[2] for r in results if r[2]])
        log(f"Date range: {dates[0]} ~ {dates[-1]}")

    # Rebuild FTS
    try:
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("""
            INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
            SELECT rowid, title, site_name, 
                   CASE WHEN length(content) > 200 THEN substr(content, 1, 200) ELSE content END
            FROM gov_raw WHERE category = ?
        """, (SOURCE,))
        conn.commit()
        conn.close()
        log("FTS index rebuilt")
    except Exception as e:
        log(f"FTS rebuild note: {e}")


if __name__ == "__main__":
    main()
