#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
讷河市 - 通知公告
CMS: 齐齐哈尔政府网站群
列表API: /search/{channelId}?_isJson=true&page=N
详情: div.article-content#zoomcon
"""

import requests
import sqlite3
import re
import time
import json
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.nehe.gov.cn"
API_URL = "https://www.nehe.gov.cn/search/c0592afb020d4bbb97a9bd7e46b1c6aa"
DB_PATH = "/root/search.db"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Referer": "https://www.nehe.gov.cn/nehe/c100473/list.shtml",
}
MAX_PAGES = 5
PAGE_SIZE = 20
SITE_NAME = "讷河市通知公告"
DELAY = 0.5

def fetch_list(page):
    params = {
        "_isAgg": "false",
        "_isJson": "true",
        "_pageSize": str(PAGE_SIZE),
        "_template": "index",
        "page": str(page),
    }
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        data = r.json()
        return data.get("data", {}).get("results", [])
    except Exception as e:
        print(f"  [ERROR] API failed: {e}")
        return []

def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] Fetch failed: {url} - {e}")
        return None

def parse_detail(soup):
    content = ""
    zoom = soup.find("div", class_="article-content", id="zoomcon")
    if zoom:
        content = str(zoom)
    if not content:
        zoom2 = soup.find("div", id="zoomcon")
        if zoom2:
            content = str(zoom2)
    title = ""
    h1 = soup.find("h1", class_="article-title")
    if h1:
        title = h1.get_text(strip=True)
    date_str = ""
    source = ""
    attr = soup.find("div", class_="article-attr")
    if attr:
        for span in attr.find_all("span"):
            txt = span.get_text(strip=True)
            if "日期" in txt:
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
                if m:
                    date_str = m.group(1)
            if "来源" in txt:
                source = txt.replace("来源：", "").strip()
    return title, content, date_str, source

def main():
    print(f"[{SITE_NAME}] 开始爬取...")
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, MAX_PAGES + 1):
        print(f"\n--- 第{page}页 ---")
        results = fetch_list(page)
        print(f"  API返回: {len(results)}条")
        if not results:
            print("  无更多数据，停止")
            break

        for item in results:
            title = item.get("title", "").strip()
            page_url = item.get("url", "").strip()
            pub_date = item.get("publishedTimeStr", "")[:10]
            source = item.get("channelName", "")
            
            if not page_url:
                continue
            
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            time.sleep(DELAY)
            detail_soup = get_soup(page_url)
            if not detail_soup:
                total_error += 1
                continue
            
            detail_title, content, det_date, det_source = parse_detail(detail_soup)
            text_len = len(re.sub(r'<[^>]+>', '', content).strip())
            if text_len < 50:
                print(f"  跳过(正文太短{text_len}): {title[:40]}")
                total_skip += 1
                continue
            
            final_title = detail_title or title
            final_date = det_date or pub_date
            final_source = det_source or source
            
            clean_text = re.sub(r'<[^>]+>', '', content).strip()
            summary = clean_text[:200]
            
            try:
                c.execute("""INSERT INTO gov_raw (site_name, title, content, page_url, source_url, publish_date, summary, status, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?, 'active', 1)""",
                          (SITE_NAME, final_title, content, page_url, final_source, final_date, summary))
                conn.commit()
                total_new += 1
                print(f"  +1 [{final_date}] {final_title[:50]}")
            except Exception as e:
                print(f"  [ERROR] DB insert: {e}")
                conn.rollback()
    
    conn.close()
    print(f"\n=== 完成 ===\n新增: {total_new}, 跳过: {total_skip}, 错误: {total_error}")

if __name__ == "__main__":
    main()
