#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
和静县人民政府 - 生态环境信息
CMS: 自定义(同前站)
列表: zwgk_list.shtml (zwgk_list_2.shtml ... zwgk_list_5.shtml)
详情: div.content_box -> h1(标题), div.tit_box(日期/来源), div.content#content(正文)
"""

import requests
import sqlite3
import re
import time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.xjhj.gov.cn"
LIST_URL = "https://www.xjhj.gov.cn/xjhjx/c117342/zwgk_list.shtml"
DB_PATH = "/root/search.db"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
MAX_PAGES = 5
SITE_NAME = "和静县生态环境信息"
DELAY = 0.5

def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] Fetch failed: {url} - {e}")
        return None

def parse_list_page(soup):
    items = []
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a.get("href", "").strip()
        # Article links contain /xjhjx/c.../ with date path
        if "/xjhjx/c" not in href or "zwgk_list" in href or "javascript" in href:
            continue
        title = a.get("title") or a.get_text(strip=True)
        if not title or len(title) < 5:
            continue
        if href.startswith("/"):
            full_url = BASE_URL + href
        elif href.startswith("http"):
            full_url = href
        else:
            full_url = urljoin(LIST_URL, href)
        date_str = ""
        span = li.find("span", class_="fr")
        if span:
            date_str = span.get_text(strip=True)
        items.append((title, full_url, date_str))
    return items

def parse_detail(soup):
    content = ""
    content_div = soup.find("div", class_="content", id="content")
    if content_div:
        content = str(content_div)
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    date_str = ""
    source = ""
    tit_box = soup.find("div", class_="tit_box")
    if tit_box:
        for span in tit_box.find_all("span"):
            txt = span.get_text(strip=True)
            m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
            if m and not date_str:
                date_str = m.group(1)
            if "来源" in txt:
                source = txt.replace("来源：", "").strip()
    return title, content, date_str, source

def main():
    print(f"[{SITE_NAME}] 开始爬取...")
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, MAX_PAGES + 1):
        url = LIST_URL.replace("zwgk_list.shtml", f"zwgk_list_{page}.shtml") if page > 1 else LIST_URL
        print(f"\n--- 第{page}页: {url}")
        soup = get_soup(url)
        if not soup:
            total_error += 1
            continue
        items = parse_list_page(soup)
        print(f"  列表项: {len(items)}")
        if not items:
            print("  无更多列表项，停止")
            break

        for title, page_url, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            time.sleep(DELAY)
            detail_soup = get_soup(page_url)
            if not detail_soup:
                total_error += 1
                continue
            detail_title, content, pub_date, source = parse_detail(detail_soup)
            text_len = len(re.sub(r'<[^>]+>', '', content).strip())
            if text_len < 50:
                print(f"  跳过(正文太短{text_len}): {title[:40]}")
                total_skip += 1
                continue
            final_title = detail_title or title
            if not pub_date:
                pub_date = list_date
            clean_text = re.sub(r'<[^>]+>', '', content).strip()
            summary = clean_text[:200]
            try:
                c.execute("""INSERT INTO gov_raw (site_name, title, content, page_url, source_url, publish_date, summary, status, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?, 'active', 1)""",
                          (SITE_NAME, final_title, content, page_url, source, pub_date[:10], summary))
                conn.commit()
                total_new += 1
                print(f"  +1 [{pub_date[:10]}] {final_title[:50]}")
            except Exception as e:
                print(f"  [ERROR] DB insert: {e}")
                conn.rollback()
    conn.close()
    print(f"\n=== 完成 ===\n新增: {total_new}, 跳过: {total_skip}, 错误: {total_error}")

if __name__ == "__main__":
    main()
