#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""爬取兰溪市 - 建设项目环境影响评价信息公示（JPAAS/JCMS）"""
import requests
import sqlite3
import re
import sys
import time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE = "http://www.lanxi.gov.cn"
API_URL = BASE + "/api-gateway/jpaas-publish-server/front/page/build/unit"
DB = "/root/search.db"
CUTOFF = datetime.now() - timedelta(days=365 * 3)
SITE_NAME = "兰溪市-建设项目环评公示"
CATEGORY = "县区"

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
HEADERS = {
    "User-Agent": UA,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

s = requests.Session()
s.headers.update(HEADERS)


def fetch_list_items():
    """通过API获取列表所有条目"""
    params = {
        "parseType": "bulidstatic",
        "webId": "3614",
        "tplSetId": "O5SnyTEouC5PsGkqGVV0p",
        "pageType": "column",
        "tagId": "新闻列表",
        "pageId": "1229857145",
    }
    r = s.get(API_URL, params=params, timeout=30,
              headers={"Referer": "http://www.lanxi.gov.cn/col/col1229857145/index.html"})
    r.raise_for_status()
    data = r.json()
    html = data["data"]["html"]
    soup = BeautifulSoup(html, "html.parser")

    items = []
    for li in soup.find_all("li"):
        a = li.find("a")
        date_div = li.find("div", class_="list-date")
        if not a or not date_div:
            continue
        href = a.get("href", "").strip()
        if not href or href == "#":
            continue
        date_str = date_div.get_text(strip=True)
        try:
            pub_date = datetime.strptime(date_str, "%Y-%m-%d")
        except:
            continue
        if pub_date < CUTOFF:
            continue
        url = href if href.startswith("http") else BASE + href
        items.append((url, date_str))
    return items


def fetch_detail(item_url):
    """获取详情页的标题和正文"""
    r = s.get(item_url, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")

    # 标题从<title>标签提取
    title_tag = soup.find("title")
    if not title_tag:
        return None, None
    title = title_tag.get_text(strip=True)
    # 去掉站点后缀
    for suffix in ["_兰溪市人民政府", " - 兰溪市人民政府", "—兰溪市人民政府"]:
        if suffix in title:
            title = title.split(suffix)[0].strip()
            break

    # 正文从 div.artic_main.main_section > div.main_section
    content_div = soup.select_one("div.artic_main.main_section > div.main_section")
    if not content_div:
        # fallback
        content_div = soup.select_one("div.main_section")
    if not content_div:
        return title, ""

    for tag in content_div.find_all(["script", "style"]):
        tag.decompose()
    content = re.sub(r"\s+", " ", str(content_div)).strip()
    return title, content


def save_to_db(items_data):
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    new_count = 0
    for title, url, date_str, content in items_data:
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if c.fetchone():
            continue
        summary = content[:300] if content else ""
        c.execute(
            "INSERT INTO gov_raw (site_name, category, title, page_url, content, summary, publish_date) VALUES (?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, CATEGORY, title, url, content, summary, date_str),
        )
        new_count += 1
    conn.commit()
    conn.close()
    return new_count


def main():
    print("[%s] 开始爬取" % SITE_NAME)

    print("  获取列表...")
    items = fetch_list_items()
    print("  找到 %d 条" % len(items))

    all_data = []
    for url, date_str in items:
        print("  详情: %s..." % url[-50:])
        title, content = fetch_detail(url)
        if not title:
            print("    [SKIP] 无标题")
            continue
        if not content:
            print("    [SKIP] 无正文")
            continue
        print("    %s | len=%d" % (title[:35], len(content)))
        all_data.append((title, url, date_str, content))
        time.sleep(0.5)

    if not all_data:
        print("无新数据")
        return

    print("\n入库 %d 条..." % len(all_data))
    n = save_to_db(all_data)
    print("新增入库: %d 条" % n)

    if n > 0:
        print("重建FTS索引...")
        conn = sqlite3.connect(DB, timeout=60)
        conn.commit()
        conn.close()

    print("完成!")


if __name__ == "__main__":
    main()
