#!/usr/bin/env python3
"""
crawl_my_gov.py - 绵阳市人民政府-公示公告爬虫
https://www.my.gov.cn/mysrmzf/c100022/list.shtml

CMS: TRS WCM，AJAX分页API
列表API: /common/search/{channelId}?_isAgg=true&_isJson=true&_pageSize=20&page={n}
总数497条，25页
详情: /mysrmzf/c100022/YYYYMM/XXXX.shtml
正文: div.article_content
"""

import os, re, sys, time, json
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "绵阳市政府-公示公告"
CATEGORY = "政府公告"
BASE = "https://www.my.gov.cn"
CHANNEL_ID = "2dc1fdddd18b48038850df9427e4402e"
API_URL = BASE + "/common/search/" + CHANNEL_ID
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
PAGE_SIZE = 20


def fetch_api(page):
    params = {
        "_isAgg": "true", "_isJson": "true",
        "_pageSize": str(PAGE_SIZE), "_template": "index",
        "_rangeTimeGte": "", "_channelName": "",
        "page": str(page),
    }
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
        return r.json() if r.status_code == 200 else None
    except:
        return None


def fetch_detail(url):
    full_url = url if url.startswith("http") else BASE + url
    try:
        r = requests.get(full_url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
    except:
        return None, "", None
    soup = BeautifulSoup(r.text, "html.parser")

    # Title
    title = ""
    h1 = soup.find("h1", class_="article_title")
    if h1:
        uc = h1.find("ucaptitle")
        if uc:
            title = uc.get_text(strip=True)
        if not title:
            title = h1.get_text(strip=True)

    # Date
    date_str = ""
    pt = soup.find("publishtime")
    if pt:
        d = pt.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{2}-\d{2})", d)
        if m:
            date_str = m.group(1)

    # Body
    body = ""
    ac = soup.find("div", class_="article_content")
    if ac:
        for t in ac.find_all(["script", "style"]):
            t.decompose()
        # Remove share buttons and other non-content elements
        for t in ac.find_all(class_=re.compile(r"(share|ewm|fav)")):
            t.decompose()
        body = str(ac).strip()

    return title, date_str, body


def main():
    conn = None
    try:
        conn = __import__("sqlite3").connect(DB_PATH, timeout=30)
        cur = conn.cursor()
    except Exception as e:
        print(f"[err] DB: {e}", flush=True)
        return

    # Get total from API
    d = fetch_api(1)
    if not d:
        print("[my] API failed!", flush=True)
        return
    total = d["data"]["total"]
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f"[my] Total: {total}, pages: {total_pages}", flush=True)

    total_new = 0
    total_skip = 0

    for page in range(1, total_pages + 1):
        d = fetch_api(page)
        if not d:
            print(f"[my] Page {page}: API failed", flush=True)
            continue

        results = d["data"].get("results", [])
        page_new = 0
        for item in results:
            url = item.get("url", "").strip()
            title_api = item.get("title", "").strip()
            pub_str = item.get("publishedTimeStr", "")
            api_date = pub_str[:10] if pub_str else ""

            if not url:
                continue

            if api_date and api_date < CUTOFF:
                continue

            full_url = url if url.startswith("http") else BASE + url

            cur.execute(
                "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
                (full_url, SITE_NAME),
            )
            if cur.fetchone():
                total_skip += 1
                continue

            # Fetch detail
            d_title, d_date, body = fetch_detail(url)
            if not d_title:
                d_title = title_api
            if not d_date:
                d_date = api_date

            date_rank = 0
            if d_date and "-" in d_date:
                try:
                    date_rank = int(d_date.replace("-", ""))
                except:
                    pass

            summary = ""
            if body:
                bs = BeautifulSoup(body, "html.parser")
                plain = bs.get_text(strip=True)
                summary = plain[:200]

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, d_title, full_url, d_date, body, date_rank, CATEGORY, summary),
                )
                if cur.rowcount > 0:
                    total_new += 1
                    page_new += 1
                    if total_new % 10 == 0:
                        conn.commit()
            except Exception as e:
                print(f"  [err] DB: {e}", flush=True)

        print(f"[my] Page {page}/{total_pages}: +{page_new} new", flush=True)
        conn.commit()
        time.sleep(0.3)

    if total_new > 0:
        try:
            cur.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            print("[my] FTS rebuilt", flush=True)
        except Exception as e:
            print(f"[my] FTS: {e}", flush=True)

    conn.close()
    print(f"[my] Done: +{total_new} new, {total_skip} skipped", flush=True)
    print(json.dumps({"site": SITE_NAME, "new": total_new, "skip": total_skip}, ensure_ascii=False))


if __name__ == "__main__":
    main()
