#!/usr/bin/env python3
"""
如皋市长江镇 - 公告公示 crawling script
Site: https://www.rugao.gov.cn/rgscjz/tzgg/tzgg.html
CMS: TrueCMS (jQuery jpage pagination)
Only page 1 accessible (30 items); pagination API behind broken dataproxy.jsp
For daily crawl, new items appear on page 1 — sufficient for incremental updates.
"""

import os
import re
import sys
import sqlite3
import requests
import json
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

# ── DB config ──────────────────────────────────────────────
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
SITE_NAME = "如皋市长江镇-公告公示"

# ── HTTP ───────────────────────────────────────────────────
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}
BASE = "https://www.rugao.gov.cn"
LIST_URL = f"{BASE}/rgscjz/tzgg/tzgg.html"

# ── DB helpers ─────────────────────────────────────────────
def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    conn.row_factory = sqlite3.Row
    return conn

def is_dup(conn, page_url):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    return cur.fetchone() is not None

def insert_item(conn, site_name, title, page_url, publish_date, content, category):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    summary = ""
    if content:
        text_soup = BeautifulSoup(content, 'html.parser')
        plain = text_soup.get_text(strip=True)
        summary = plain[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (site_name, title, page_url, publish_date, content, date_rank, category, summary)
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"  [ERR] insert: {e}")
        return False

# ── Parse list page ────────────────────────────────────────
def parse_list_page(html):
    """Parse initData XML from page HTML, extract (uuid, title, url, date) tuples."""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    init_div = soup.find(id="initData")
    if init_div:
        uls = init_div.find_all("ul", class_="list-ul")
    else:
        uls = soup.find_all("ul", class_="list-ul")

    for ul in uls:
        li = ul.find("li")
        if not li:
            continue
        a = li.find("a")
        span = li.find("span")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get_text(strip=True)
        date_str = span.get_text(strip=True)
        if not href or not title:
            continue
        # Extract UUID from URL
        m = re.search(r'/([a-f0-9-]{36})\.html', href)
        if not m:
            continue
        url = href if href.startswith("http") else BASE + href
        items.append((m.group(1), title, url, date_str))

    return items

# ── Parse detail page ──────────────────────────────────────
def parse_detail(html):
    """Extract (title, content, publish_date) from detail page."""
    soup = BeautifulSoup(html, "html.parser")

    # Title from <div class="container-main-title">
    title_div = soup.find("div", class_="container-main-title")
    title = title_div.get_text(strip=True) if title_div else ""

    # Content from <div id="zoom" class="test-1">
    zoom = soup.find("div", id="zoom", class_="test-1")
    content = ""
    if zoom:
        # Keep full HTML for attachments/links
        content = str(zoom)

    # Date from <meta name="PubDate">
    pub_date = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        pub_date = meta["content"].strip()
        m2 = re.match(r"(\d{4}-\d{1,2}-\d{1,2})", pub_date)
        if m2:
            pub_date = m2.group(1)

    if not title:
        meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_t and meta_t.get("content"):
            title = re.sub(r'<[^>]+>', '', meta_t["content"]).strip()

    return title, content, pub_date

# ── Main ───────────────────────────────────────────────────
def main():
    conn = get_conn()

    # 1. Fetch list page
    print(f"[{SITE_NAME}] Fetching list page: {LIST_URL}")
    resp = requests.get(LIST_URL, headers=HEADERS, timeout=30, verify=False)
    resp.encoding = "utf-8"
    html = resp.text

    items = parse_list_page(html)
    print(f"[{SITE_NAME}] Found {len(items)} items on page 1")

    # 2. Filter by cutoff date
    active = [(u, t, url, d) for u, t, url, d in items if d >= CUTOFF_DATE]
    skipped = [(u, t, d) for u, t, _, d in items if d < CUTOFF_DATE]
    print(f"[{SITE_NAME}] Within cutoff ({CUTOFF_DATE}): {len(active)} items")
    if skipped:
        print(f"[{SITE_NAME}] Skipped (older): {len(skipped)}")
        for u, t, d in skipped:
            print(f"  SKIP {d} {t[:50]}")

    # 3. Fetch and save detail pages
    new_count = 0
    dup_count = 0
    error_count = 0

    for idx, (uuid, title, url, date_str) in enumerate(active, 1):
        try:
            # Check duplicate via page_url
            if is_dup(conn, url):
                dup_count += 1
                continue

            detail_resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            detail_resp.encoding = "utf-8"
            detail_html = detail_resp.text

            detail_title, content, pub_date = parse_detail(detail_html)
            if not detail_title:
                detail_title = title
            if not pub_date:
                pub_date = date_str

            inserted = insert_item(conn, SITE_NAME, detail_title, url, pub_date, content, "环保公示")
            if inserted:
                new_count += 1
            else:
                dup_count += 1

        except Exception as e:
            error_count += 1
            print(f"  ERROR [{uuid[:8]}] {title[:40]}: {e}")

        conn.commit()

        if idx % 10 == 0:
            print(f"  Progress: {idx}/{len(active)} (new={new_count}, dup={dup_count}, err={error_count})")

    conn.close()

    print(f"[{SITE_NAME}] Done!")
    print(f"  Total: {len(active)} | New: {new_count} | Duplicate: {dup_count} | Error: {error_count}")
    return {"site": SITE_NAME, "total": len(active), "new": new_count, "dup": dup_count, "err": error_count}


if __name__ == "__main__":
    result = main()
    print(json.dumps(result, ensure_ascii=False))
