#!/usr/bin/env python3
"""
crawl_bynesrmtzx.py - 巴彦淖尔新闻网-通知公告爬虫
https://www.bynesrmtzx.cn/col38.html

CMS: 自定义CMS
列表: col38.html (首页), col38_2.html ~ col38_14.html (14页, ~25条/页)
详情: /content/YYYY/MM/DD/cXXXXX.html
正文: div.detail_main (含 <!--enpcontent-->)
"""

import os, re, sys, time, json
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "bynesrmtzx.cn-通知公告"
CATEGORY = "政府公告"
BASE_URL = "https://www.bynesrmtzx.cn"
LIST_URL = BASE_URL + "/col38.html"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
TOTAL_PAGES = 14  # confirmed: col38.html ~ col38_14.html, 15=404


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text if r.status_code == 200 else ""
    except:
        return ""


def parse_detail(url):
    """Extract title, date, body from detail page."""
    html = fetch(url)
    if not html:
        return None, None, ""
    soup = BeautifulSoup(html, "html.parser")

    # Title
    title = ""
    dt = soup.find("div", class_="detail_title")
    if dt:
        title = dt.get_text(strip=True)

    # Date
    date_str = ""
    dm = soup.find("div", class_="detail_message")
    if dm:
        spans = dm.find_all("span")
        for s in spans:
            txt = s.get_text(strip=True)
            m = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
            if m:
                date_str = m.group(1)
                break

    # Body - div.detail_main
    body = ""
    dm2 = soup.find("div", class_="detail_main")
    if dm2:
        for t in dm2.find_all(["script", "style"]):
            t.decompose()
        body = str(dm2).strip()

    return title, date_str, body


def main():
    conn = None
    try:
        conn = __import__("sqlite3").connect(DB_PATH, timeout=30)
        cur = conn.cursor()
    except Exception as e:
        print(f"[err] DB connect failed: {e}", flush=True)
        return

    total_new = 0
    total_skip = 0

    for page in range(1, TOTAL_PAGES + 1):
        if page == 1:
            list_url = LIST_URL
        else:
            list_url = BASE_URL + f"/col38_{page}.html"

        html = fetch(list_url)
        if not html:
            print(f"[byn] Page {page}/{TOTAL_PAGES}: fetch failed", flush=True)
            continue

        soup = BeautifulSoup(html, "html.parser")
        links = soup.select("a.news-list-link")
        if not links:
            print(f"[byn] Page {page}/{TOTAL_PAGES}: no items", flush=True)
            continue

        page_new = 0
        for a in links:
            href = a.get("href", "")
            if not href:
                continue
            if href.startswith("/"):
                detail_url = BASE_URL + href
            elif href.startswith("http"):
                detail_url = href
            else:
                detail_url = BASE_URL + "/" + href

            # Get date from list
            li = a.find("li")
            date_el = li.find("div", class_="date") if li else None
            list_date = date_el.get_text(strip=True) if date_el else ""

            if list_date and list_date < CUTOFF:
                continue

            # Check existing
            cur.execute(
                "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
                (detail_url, SITE_NAME),
            )
            if cur.fetchone():
                total_skip += 1
                continue

            # Fetch detail
            d_title, d_date, body = parse_detail(detail_url)
            if not d_title:
                d_title = a.get_text(strip=True) or "无标题"
            if not d_date:
                d_date = list_date

            date_rank = 0
            if d_date and "-" in d_date:
                try:
                    date_rank = int(d_date.replace("-", ""))
                except:
                    pass

            summary = ""
            if body:
                bs = BeautifulSoup(body, "html.parser")
                plain = bs.get_text(strip=True)
                summary = plain[:200]

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary),
                )
                if cur.rowcount > 0:
                    total_new += 1
                    page_new += 1
                    if total_new % 20 == 0:
                        conn.commit()
            except Exception as e:
                print(f"  [err] DB: {e}", flush=True)

        print(
            f"[byn] Page {page}/{TOTAL_PAGES}: +{page_new} new",
            flush=True,
        )
        conn.commit()
        time.sleep(0.5)

    # FTS rebuild
    if total_new > 0:
        try:
            cur.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            print("[byn] FTS rebuilt", flush=True)
        except Exception as e:
            print(f"[byn] FTS: {e}", flush=True)

    conn.close()
    print(f"[byn] Done: +{total_new} new, {total_skip} existing skipped", flush=True)
    print(json.dumps({"site": SITE_NAME, "new": total_new, "skip": total_skip}, ensure_ascii=False))


if __name__ == "__main__":
    main()
