#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
宁波市北仑区 - 建设项目环境影响评价信息公示
CMS: Hanweb JPAAS
列表: JPAAS API -> pageNo参数
详情: div.bt-content-news#zoom
"""

import requests
import sqlite3
import re
import time
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "https://www.bl.gov.cn"
API_URL = "https://www.bl.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
DB_PATH = "/root/search.db"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://www.bl.gov.cn/col/col1229893313/index.html",
}
SITE_NAME = "宁波北仑区环评信息公示"
DELAY = 0.5

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "3499",
    "tplSetId": "iIEzq5wa84hjU6xuCoM3q",
    "pageType": "column",
    "tagId": "政务公开信息列表",
    "editType": "null",
    "pageId": "1229893313",
}

def get_list_api(page):
    params = API_PARAMS.copy()
    param_json = '{"pageNo":%d,"pageSize":15}' % page
    params["paramJson"] = param_json
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        return r.json()
    except Exception as e:
        print(f"  [ERROR] API failed: {e}")
        return None

def parse_list_from_api(data):
    items = []
    html = data.get("data", {}).get("html", "")
    if not html:
        return items
    # Extract links
    pattern = r'<a[^>]+href="([^"]+)"[^>]+title="([^"]*)"[^>]*>(.*?)</a>'
    matches = re.findall(pattern, html, re.DOTALL)
    for href, title_in_attr, link_text in matches:
        # Skip pagination buttons
        if href.startswith("#") or title_in_attr in ("首页", "上页", "下页", "尾页"):
            continue
        if not title_in_attr:
            title = re.sub(r'<[^>]+>', '', link_text).strip()
        else:
            title = title_in_attr.strip()
        if not title:
            continue
        # URL
        if href.startswith("/"):
            url = BASE_URL + href
        elif href.startswith("http"):
            url = href
        else:
            url = BASE_URL + "/" + href
        # Date
        date_match = re.search(r'<span[^>]*>\[(\d{4}-\d{2}-\d{2})\]</span>', html[html.find(href):html.find(href)+500])
        date_str = date_match.group(1) if date_match else ""
        items.append((title, url, date_str))
    return items

def parse_detail(soup):
    content = ""
    zoom = soup.find("div", class_="bt-content-news", id="zoom")
    if zoom:
        content = str(zoom)
    if not content:
        zoom2 = soup.find("div", id="zoom")
        if zoom2:
            content = str(zoom2)
    return content.strip()

def get_meta(soup):
    title = ""
    t = soup.find("div", class_="bt-news-detail-title")
    if t:
        title = t.get_text(strip=True)
    date_str = ""
    contip = soup.find("div", class_="contip")
    if contip:
        spans = contip.find_all("span")
        for span in spans:
            txt = span.get_text(strip=True)
            m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
            if m:
                date_str = m.group(1)
                break
    source = ""
    meta_src = soup.find("meta", attrs={"name": "ContentSource"})
    if meta_src:
        source = meta_src.get("content", "")
    return title, date_str, source

def main():
    print(f"[{SITE_NAME}] 开始爬取...")
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    total_error = 0

    # Page 1 first
    for page in [1]:
        print(f"\n--- 第{page}页 ---")
        data = get_list_api(page)
        if not data:
            total_error += 1
            continue
        items = parse_list_from_api(data)
        print(f"  列表项: {len(items)}")
        if not items:
            break

        for title, page_url, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue

            time.sleep(DELAY)
            try:
                r = requests.get(page_url, headers=HEADERS, timeout=20, verify=False)
                r.encoding = "utf-8"
                detail_soup = BeautifulSoup(r.text, "html.parser")
            except Exception as e:
                print(f"  [ERROR] Fetch detail: {e}")
                total_error += 1
                continue

            content = parse_detail(detail_soup)
            text_len = len(re.sub(r'<[^>]+>', '', content).strip())
            if text_len < 50:
                print(f"  跳过(正文太短{text_len}): {title[:40]}")
                total_skip += 1
                continue

            detail_title, pub_date, source = get_meta(detail_soup)
            final_title = detail_title or title
            if not pub_date:
                pub_date = list_date

            clean_text = re.sub(r'<[^>]+>', '', content).strip()
            summary = clean_text[:200]

            try:
                c.execute("""
                    INSERT INTO gov_raw (site_name, title, content, page_url, source_url, publish_date, summary, status, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?, 'active', 1)
                """, (SITE_NAME, final_title, content, page_url, source, pub_date[:10], summary))
                conn.commit()
                total_new += 1
                print(f"  +1 [{pub_date[:10]}] {final_title[:50]}")
            except Exception as e:
                print(f"  [ERROR] DB insert: {e}")
                conn.rollback()

    conn.close()
    print(f"\n=== 完成 ===")
    print(f"新增: {total_new}, 跳过: {total_skip}, 错误: {total_error}")

if __name__ == "__main__":
    main()
