#!/usr/bin/env python3
"""
crawl_tongbai.py — 桐柏县人民政府·通知公告
============================================
安棚镇信息公开，TRS CMS（南阳统一站点平台）
首页静态HTML，分页AJAX加载（JS渲染，curl无法获取）

列表：<li><a href="https://www.tongbai.gov.cn/YYYY/MM-DD/XXXXXXX.html" title="TITLE">TITLE</a><b>YYYY-MM-DD</b></li>
详情：<div class="content" id="content">正文 | <meta name="PubDate" content="2026/05-26">

用法:
    python3 crawl_tongbai.py             # 全量
    python3 crawl_tongbai.py 1           # 增量（只爬首页）
"""

import os
import re
import sys
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

# ── DB ──────────────────────────────────────────────
DB_PATH = os.environ.get("GOV_DB_PATH", os.path.expanduser(
    "~/Nutstore Files/我的坚果云/Crawler/gov_crawler/search.db"
))

# ── URLs ────────────────────────────────────────────
BASE_URL = "https://www.tongbai.gov.cn"
LIST_URL = BASE_URL + "/apzrmzf/dtxx/tzgg/"
SITE_NAME = "桐柏县人民政府"
SOURCE = "桐柏县通知公告"

# ── Headers ─────────────────────────────────────────
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
}

# ── help ────────────────────────────────────────────
def log(msg):
    print(f"[tongbai] {msg}")

# ── Clean title ──────────────────────────────────────
def clean_title(title):
    """Remove site suffix from title"""
    return title.replace("-通知公告-桐柏县人民政府", "").replace(
        "-通知公告", ""
    ).replace("-桐柏县人民政府", "").strip()

# ── Fetch list page ──────────────────────────────────
def fetch_list():
    """Fetch page 1 list, return [(url, title, date_str), ...]"""
    log(f"Fetching list: {LIST_URL}")
    resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
    resp.encoding = "utf-8"
    html = resp.text

    soup = BeautifulSoup(html, "html.parser")
    items = []

    # The list is in <div class="zfxxgk_zdgkc"><ul><li>...</li>
    list_div = soup.find("div", class_="zfxxgk_zdgkc")
    if not list_div:
        log("ERROR: cannot find list container")
        return items

    ul = list_div.find("ul")
    if not ul:
        log("ERROR: cannot find list <ul>")
        return items

    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        href = a["href"].strip()
        # Skip weixin links
        if "mp.weixin.qq.com" in href or not href.startswith(BASE_URL):
            continue
        title = a.get("title", "") or a.get_text(strip=True)
        # Get date from <b> tag
        b = li.find("b")
        date_str = b.get_text(strip=True) if b else ""
        items.append((href, title, date_str))

    log(f"Found {len(items)} articles on page 1")
    return items

# ── Fetch detail ─────────────────────────────────────
def fetch_detail(url):
    """Return (content_html, date_str) or (None, None)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"  ERROR fetching {url}: {e}")
        return None, None

    soup = BeautifulSoup(html, "html.parser")

    # Content: <div class="content" id="content">
    content_div = soup.find("div", class_="content", id="content")
    content_html = ""
    if content_div:
        content_html = str(content_div)

    # Date: <meta name="PubDate" content="2026/05-26">
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        raw = meta["content"].strip()
        # Format like "2026/05-26" -> "2026-05-26"
        raw = raw.replace("/", "-")
        # Try to parse as date
        try:
            datetime.strptime(raw, "%Y-%m-%d")
            date_str = raw
        except ValueError:
            pass

    # Fallback: from visible text "时间：YYYY-MM-DD"
    if not date_str:
        m = re.search(r'时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            date_str = m.group(1)

    # Fallback: from list date embedded in URL
    if not date_str:
        m = re.search(r'/(\d{4})/(\d{2})-(\d{2})/', url)
        if m:
            date_str = f"{m.group(1)}-{m.group(2)}-{m.group(3)}"

    if not content_html:
        log(f"  WARNING: no content for {url}")
        return None, date_str

    return content_html, date_str

# ── Save to DB ───────────────────────────────────────
def save_to_db(items):
    """Upsert records into search.db"""
    import sqlite3

    if not items:
        log("No items to save")
        return

    log(f"Opening DB: {DB_PATH}")
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()

    inserted = 0
    for url, title, date_str, content_html in items:
        try:
            c.execute(
                """INSERT OR REPLACE INTO gov_raw
                   (site_name, page_url, title, publish_date, content, category)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (
                    SITE_NAME,
                    url,
                    clean_title(title),
                    date_str,
                    content_html,
                    SOURCE,
                ),
            )
            inserted += 1
        except Exception as e:
            log(f"  DB error for {url}: {e}")

    conn.commit()
    conn.close()
    log(f"Saved {inserted}/{len(items)} records to DB")

# ── Main ────────────────────────────────────────────
def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] == "1"

    # 3-year cutoff
    three_years_ago = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
    log(f"3-year cutoff: {three_years_ago}")

    # Fetch list
    articles = fetch_list()
    if not articles:
        log("No articles found, exiting")
        return

    # Filter: 3-year
    articles_filtered = [(u, t, d) for u, t, d in articles if d and d >= three_years_ago]
    log(f"After 3-year filter: {len(articles_filtered)}/{len(articles)}")

    if incremental:
        # Incremental: only crawl the first (most recent) page
        log("Incremental mode: crawling page 1 only")
        to_process = articles_filtered
    else:
        to_process = articles_filtered
        log(f"Full mode: crawling {len(to_process)} articles")

    # Fetch details
    results = []
    for i, (url, title, date_str) in enumerate(to_process, 1):
        log(f"[{i}/{len(to_process)}] {title[:50]}...")
        content_html, detail_date = fetch_detail(url)
        # Use detail_date if available, else fallback to list date
        final_date = detail_date or date_str
        if content_html:
            results.append((url, title, final_date, content_html))
        else:
            # Save even without content (for record keeping)
            results.append((url, title, final_date, ""))

    # Save
    save_to_db(results)

    # Summary
    log(f"\n{'='*50}")
    log(f"Done. Total: {len(results)} articles")
    if results:
        dates = sorted([r[2] for r in results if r[2]])
        log(f"Date range: {dates[0]} ~ {dates[-1]}")
    log(f"{'='*50}")


if __name__ == "__main__":
    main()
