#!/usr/bin/env python3
"""
新星经济技术开发区管理委员会 - 公示公告
https://www.btnsss.gov.cn/info/iList.jsp?node_id=GKxxs&isSd=false&cat_id=12053
Custom CMS, jQuery createPage pagination, 216 pages
"""
import os, sys, re, requests, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "新星经济技术开发区-公示公告"
BASE_URL = "https://www.btnsss.gov.cn"
LIST_PATH = "/info/iList.jsp?isSd=false&node_id=GKxxs&cat_id=12053"
THREADS = 10
THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
TOTAL_PAGES = 216  # pageCount=216

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

def get_list_url(page):
    """获取列表页URL (page从1开始)"""
    if page == 1:
        return BASE_URL + LIST_PATH
    return BASE_URL + LIST_PATH + f"&cur_page={page}"

def parse_list(html):
    """解析列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("ul.gk-nr-ul li"):
        a = li.find("a", class_="gk-br")
        span = li.find("span")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get_text(strip=True).lstrip("•").strip()
        date = span.get_text(strip=True)[:10]
        if not href or not title:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append({"title": title, "url": href, "date": date})
    return items

def fetch_detail(url):
    """抓取详情页"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        soup = BeautifulSoup(r.text, "html.parser")

        # 标题
        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()

        # 日期
        publish_date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            pub = meta_d["content"].strip()
            publish_date = pub.split(" ")[0] if " " in pub else pub[:10]

        # 正文
        content_div = soup.select_one("div.con-nr")
        content = str(content_div) if content_div else ""

        if not content or len(content.strip()) < 30:
            return None

        return {
            "title": title,
            "date": publish_date,
            "content": content,
        }
    except Exception as e:
        print(f"  [!] 详情失败: {os.path.basename(url)} - {e}", file=sys.stderr)
        return None

def store_item(item):
    """写入search.db"""
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, publish_date, site_name, page_url, content, summary)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (
            item["title"], item["date"], SITE_NAME, item["url"],
            item.get("content", ""), item.get("summary", "")
        ))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected > 0
    except Exception as e:
        print(f"  [!] DB写入失败: {e}", file=sys.stderr)
        return False

def main():
    print(f"[*] 站点: {SITE_NAME}")
    print(f"[*] 时间阈值: {THRESHOLD}")
    print(f"[*] 总页数: {TOTAL_PAGES}")

    # 1. 抓取所有列表页
    all_items = []
    for page in range(1, TOTAL_PAGES + 1):
        url = get_list_url(page)
        try:
            r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
            r.encoding = "utf-8"
            if r.status_code != 200:
                if page % 10 == 1:
                    print(f"  [!] 列表页 {page} 失败: HTTP {r.status_code}")
                continue
            items = parse_list(r.text)
            if not items:
                print(f"  [!] 列表页 {page} 无数据")
                continue
            all_items.extend(items)
            if page % 20 == 0:
                print(f"  [→] 列表页 {page}/{TOTAL_PAGES}: 累计 {len(all_items)} 条")
        except Exception as e:
            if page % 10 == 1:
                print(f"  [!] 列表页 {page} 异常: {e}")

    print(f"\n[*] 共获取 {len(all_items)} 条")

    # 2. 过滤近3年
    items_to_fetch = [it for it in all_items if it["date"] >= THRESHOLD]
    print(f"[*] 3年内({THRESHOLD}~): {len(items_to_fetch)} 条")
    print(f"[*] 3年外: {len(all_items) - len(items_to_fetch)} 条")

    if not items_to_fetch:
        print("[*] 无需新增")
        return

    # 3. 去重
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        existing = set()
        for it in items_to_fetch:
            c = conn.cursor()
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], SITE_NAME))
            if c.fetchone():
                existing.add(it["url"])
        conn.close()
    except Exception as e:
        print(f"  [!] 去重失败: {e}")
        existing = set()

    to_crawl = [it for it in items_to_fetch if it["url"] not in existing]
    print(f"[*] 需爬详情: {len(to_crawl)} 条 (已有 {len(items_to_fetch)-len(to_crawl)} 条跳过)")

    if not to_crawl:
        print("[*] 无需新增")
        return

    # 4. 并发爬详情
    new_count = 0
    skip_count = 0

    with ThreadPoolExecutor(max_workers=THREADS) as executor:
        fut_map = {executor.submit(fetch_detail, it["url"]): it for it in to_crawl}
        for fut in as_completed(fut_map):
            it = fut_map[fut]
            try:
                detail = fut.result()
                if detail:
                    it["content"] = detail["content"]
                    it["title"] = detail.get("title", it["title"])
                    it["date"] = detail.get("date", it["date"])
                    if store_item(it):
                        new_count += 1
                        if new_count % 20 == 0:
                            print(f"  [✓] #{new_count} {it['title'][:35]}")
                    else:
                        skip_count += 1
                else:
                    skip_count += 1
            except Exception as e:
                print(f"  [!] 异常: {e}")
                skip_count += 1

    print(f"\n{'='*50}")
    print(f"  新增: {new_count}")
    print(f"  跳过(已存在/无正文): {skip_count}")
    print(f"  总计: {len(to_crawl)}")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
