#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Crawler for 莆田市秀屿区人民政府 - 公示公告 (ptxy.gov.cn)
CMS: 福建21pt CMS (Avalon JS)
List: div.gl_list > ul > li > span.bf-pass(date) + a[href][title]
      全部78条渲染在首页静态HTML中，无需分页
Detail: div.xl_con1#detailCont > div.TRS_Editor > p/table
Title: meta ArticleTitle
Date: meta PubDate
Attachments: a[href*=/uploadfiles/] or a[href*=.doc/.pdf/.xls]

Usage:
    python3 /root/gov_crawler/crawl_ptxy_gsgg.py
"""

import requests, re, json, sqlite3, time, os, sys
from datetime import datetime
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "莆田市秀屿区-公示公告"
CATEGORY = "通知公告"
GROUP = "福建"
LIST_URL = "http://www.ptxy.gov.cn/zwgk/gsgg/"
BASE_URL = "http://www.ptxy.gov.cn"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


# ---------- List ----------

def extract_all_items():
    try:
        r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  Error fetching list: {e}", file=sys.stderr)
        return []

    items = []
    # Find all list containers (5 chunks of ~15 items each)
    for gl_list in soup.find_all("div", class_="gl_list"):
        for li in gl_list.find_all("li"):
            a = li.find("a", href=True)
            if not a:
                continue
            href = a["href"]
            page_url = urljoin(LIST_URL, href)
            title = a.get("title", "").strip() or a.get_text(strip=True)
            if not title:
                continue
            span = li.find("span", class_="bf-pass")
            date_str = span.get_text(strip=True) if span else ""
            items.append((title, page_url, date_str))

    return items


# ---------- Detail ----------

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  fetch_detail error {url}: {e}", file=sys.stderr)
        return None, None, None, []

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title ---
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    # --- Date ---
    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        date_str = meta_date["content"].strip()[:10]

    # --- Content ---
    content = ""
    detail_cont = soup.find("div", id="detailCont")
    if detail_cont:
        # Find TRS_Editor inside
        editor = detail_cont.find("div", class_="TRS_Editor")
        if not editor:
            editor = detail_cont  # Fallback to entire detail container
        for s in editor.find_all(["script", "style"]):
            s.decompose()
        parts = []
        # Buffer for inline spans (text fragments)
        text_buffer = []
        for child in editor.children:
            if not child.name:
                continue
            if child.name == "table":
                text_joined = "".join(text_buffer).strip()
                if text_joined:
                    parts.append(text_joined)
                text_buffer = []
                parts.append(str(child))
            elif child.name == "p":
                # Flush inline buffer first
                text_joined = "".join(text_buffer).strip()
                if text_joined:
                    parts.append(text_joined)
                    text_buffer = []
                # Skip p inside table cells (already as table HTML)
                if child.find_parent("td"):
                    continue
                t = child.get_text(strip=True)
                if t and t not in (" ", "\xa0", "&nbsp;"):
                    parts.append(t)
            else:
                # span, div, font etc. - accumulate
                if child.find_parent("td"):
                    continue
                t = child.get_text(strip=True)
                if t and t not in (" ", "\xa0", "&nbsp;"):
                    text_buffer.append(t)
        # Flush remaining buffer
        text_joined = "".join(text_buffer).strip()
        if text_joined:
            parts.append(text_joined)
        content = "\n\n".join(parts)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()

    # --- Attachments ---
    attachments = []
    seen = set()
    for a_tag in soup.find_all("a", href=True):
        href = a_tag["href"]
        is_attachment = (
            re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt)$", href, re.I) or
            "/uploadfiles/" in href or
            "/upload/" in href
        )
        if is_attachment:
            full_url = urljoin(url, href)
            name = a_tag.get_text(strip=True) or href.split("/")[-1].split("?")[0]
            if full_url not in seen:
                seen.add(full_url)
                attachments.append({"name": name, "url": full_url})

    return title, date_str, content, attachments


# ---------- DB ----------

def with_retry(fn, desc="DB op", max_attempts=10, delay=5):
    for attempt in range(1, max_attempts + 1):
        try:
            return fn()
        except sqlite3.OperationalError as e:
            if "locked" in str(e) and attempt < max_attempts:
                print(f"  {desc}: locked (attempt {attempt}/{max_attempts}), retry in {delay}s...", file=sys.stderr)
                time.sleep(delay)
            else:
                print(f"  {desc} failed: {e}", file=sys.stderr)
                return False
    return False


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn


def save_article(conn, title, page_url, publish_date, content, attachments):
    summary = content[:200] if content else title
    summary = re.sub(r"\s+", " ", summary).strip()
    att_json = json.dumps(attachments, ensure_ascii=False) if attachments else "[]"
    try:
        cur = conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, page_url, page_url, title.strip(), publish_date,
             summary, content, CATEGORY, att_json, GROUP),
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"  DB error: {e}", file=sys.stderr)
        return False


# ---------- Main ----------

def main():
    print(f"Scraping {SITE_NAME}", file=sys.stderr)
    conn = init_db()

    def do_clean():
        conn.execute("DELETE FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        conn.commit()
    print("  Cleaning old data...", file=sys.stderr)
    with_retry(do_clean, desc=f"Clean {SITE_NAME}")
    print(f"  Cleaned old data for {SITE_NAME}", file=sys.stderr)

    all_items = extract_all_items()
    total_found = len(all_items)
    total_new = 0
    start_time = time.time()

    print(f"  Found {total_found} items total", file=sys.stderr)

    for title, page_url, date in all_items:
        cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
        if cur.fetchone():
            continue

        det_title, det_date, det_content, det_atts = fetch_detail(page_url)
        final_title = det_title or title
        final_date = det_date or date
        final_content = det_content or f"[{final_title}]({page_url})"
        final_atts = det_atts or []

        if save_article(conn, final_title, page_url, final_date, final_content, final_atts):
            total_new += 1

        time.sleep(0.3)

    conn.commit()
    elapsed = int(time.time() - start_time)
    conn.close()
    print(f"\nDone! Found: {total_found}, New: {total_new}, Time: {elapsed//60}m{elapsed%60:02d}s", file=sys.stderr)


if __name__ == "__main__":
    main()
