#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
第十三师新星市 - 公示公告 (cat_id=10161)
https://www.btnsss.gov.cn/info/iList.jsp?node_id=GKxxs&isSd=false&cat_id=10161
"""
import sys, os, re, time, json, requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DOMAIN = "https://www.btnsss.gov.cn"
LIST_URL = "https://www.btnsss.gov.cn/info/iList.jsp?node_id=GKxxs&isSd=false&cat_id=10161"
SITE_NAME = "第十三师新星市-公示公告"
MAX_PAGES = 3

def fetch(url, encoding="utf-8"):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = encoding
        return r.text
    except Exception as e:
        print("  FETCH ERROR %s: %s" % (url, e), file=sys.stderr)
        return None

def parse_list_items(html):
    items = []
    soup = BeautifulSoup(html, "lxml")
    for ul in soup.find_all("ul"):
        for li in ul.find_all("li"):
            a = li.find("a", class_="gk-br")
            sp = li.find("span")
            if not a or not sp or not a.get("href"):
                continue
            href = a["href"].strip()
            if not href.startswith("/gk/"):
                continue
            title = a.get_text(strip=True)
            title = re.sub(r"^[•·\s]+", "", title).strip()
            if not title or len(title) < 5:
                continue
            date = sp.get_text(strip=True)[:10]
            items.append({"url": DOMAIN + href, "title": title, "date": date})
    return items

def get_content(url, title_from_list):
    html = fetch(url)
    if not html:
        return "", "", [], title_from_list
    soup = BeautifulSoup(html, "lxml")
    full_title = title_from_list
    con_tt = soup.select_one("div.con-tt")
    if con_tt:
        t = con_tt.get_text(strip=True)
        if t:
            full_title = t
    pub_date = ""
    file_table = soup.select_one("div.file-table")
    if file_table:
        m = re.search(r"发布日期.*?<td[^>]*>(\d{4}[-/]\d{2}[-/]\d{2})", str(file_table), re.DOTALL)
        if m:
            pub_date = m.group(1).replace("/", "-")
    # 正文
    content = ""
    con_div = soup.select_one("div.con-nr")
    if not con_div:
        con_div = soup.select_one("div.con-txt")
    if con_div:
        content = str(con_div)
    return full_title, pub_date, [], content

def run(incremental=False):
    from datetime import datetime, timedelta
    cutoff = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
    records = []
    seen = set()
    max_pages = 1 if incremental else MAX_PAGES
    for pg in range(1, max_pages + 1):
        url = LIST_URL + ("&pageNo=%d" % pg if pg > 1 else "")
        html = fetch(url)
        if not html:
            continue
        items = parse_list_items(html)
        if not items:
            break
        print(f"  Page {pg}: {len(items)} items")
        for it in items:
            if it["date"] and it["date"] < cutoff:
                continue
            if it["url"] in seen:
                continue
            seen.add(it["url"])
            title, pub_date, _, content = get_content(it["url"], it["title"])
            if not title or not content.strip():
                continue
            records.append({
                "title": title,
                "url": it["url"],
                "pub_date": pub_date or it["date"],
                "site_name": SITE_NAME,
                "content": content,
                "summary": "",
            })
    print(f"  Total: {len(records)} items")
    valid = [r for r in records if r["content"].strip()]
    if valid:
        push_to_searchdb(valid, "btnsss_10161")
    return len(valid)

if __name__ == "__main__":
    cnt = run(incremental="--incremental" in sys.argv)
    print(f"Done: {cnt} records")
