#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
和静县人民政府 - 意见征集
CMS: 自定义
列表: https://www.xjhj.gov.cn/xjhjx/c110841/list.shtml (list_2.shtml ... list_9.shtml)
详情: div.content_box -> h1(标题), div.tit_box(日期/来源), div.content#content(正文)
"""

import requests
import sqlite3
import re
import time
from datetime import datetime
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.xjhj.gov.cn"
LIST_URL = "https://www.xjhj.gov.cn/xjhjx/c110841/list.shtml"
DB_PATH = "/root/search.db"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 5
SITE_NAME = "和静县意见征集"
DELAY = 0.5

def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] Fetch failed: {url} - {e}")
        return None

def parse_list_page(soup):
    """从列表页解析条目"""
    items = []
    # Find <li> items that contain <a href="/xjhjx/c110841/...">
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a.get("href", "").strip()
        # Only items with /c110841/ in URL are real articles
        if "/c110841/" not in href:
            continue
        
        title = a.get("title") or a.get_text(strip=True)
        if not title or len(title) < 5:
            continue
        
        if href.startswith("/"):
            full_url = BASE_URL + href
        elif href.startswith("http"):
            full_url = href
        else:
            full_url = urljoin(LIST_URL, href)
        
        date_str = ""
        span = li.find("span", class_="fr")
        if span:
            date_str = span.get_text(strip=True)
        
        items.append((title, full_url, date_str))
    
    return items

def parse_detail(soup):
    """解析详情页"""
    # Content
    content = ""
    content_div = soup.find("div", class_="content", id="content")
    if content_div:
        content = str(content_div)
    if not content:
        content_div = soup.find("div", id="content")
        if content_div:
            content = str(content_div)
    
    # Title
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    
    # Date and source
    date_str = ""
    source = ""
    tit_box = soup.find("div", class_="tit_box")
    if tit_box:
        spans = tit_box.find_all("span")
        for span in spans:
            txt = span.get_text(strip=True)
            if "发布日期" in txt:
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
                if m:
                    date_str = m.group(1)
            elif "来源" in txt:
                source = txt.replace("来源：", "").strip()
    
    return title, content, date_str, source

def main():
    print(f"[{SITE_NAME}] 开始爬取...")
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = LIST_URL.replace("list.shtml", f"list_{page}.shtml")
        
        print(f"\n--- 第{page}页: {url}")
        soup = get_soup(url)
        if not soup:
            total_error += 1
            continue
        
        items = parse_list_page(soup)
        print(f"  列表项: {len(items)}")
        
        if not items:
            print("  无更多列表项，停止")
            break
        
        for title, page_url, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            
            time.sleep(DELAY)
            detail_soup = get_soup(page_url)
            if not detail_soup:
                total_error += 1
                continue
            
            detail_title, content, pub_date, source = parse_detail(detail_soup)
            text_len = len(re.sub(r'<[^>]+>', '', content).strip())
            if text_len < 50:
                print(f"  跳过(正文太短{text_len}): {title[:40]}")
                total_skip += 1
                continue
            
            final_title = detail_title or title
            if not pub_date:
                pub_date = list_date
            
            clean_text = re.sub(r'<[^>]+>', '', content).strip()
            summary = clean_text[:200]
            
            try:
                c.execute("""
                    INSERT INTO gov_raw (site_name, title, content, page_url, source_url, publish_date, summary, status, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?, 'active', 1)
                """, (SITE_NAME, final_title, content, page_url, source, pub_date[:10], summary))
                conn.commit()
                total_new += 1
                print(f"  +1 [{pub_date[:10]}] {final_title[:50]}")
            except Exception as e:
                print(f"  [ERROR] DB insert: {e}")
                conn.rollback()
    
    conn.close()
    print(f"\n=== 完成 ===")
    print(f"新增: {total_new}, 跳过: {total_skip}, 错误: {total_error}")

if __name__ == "__main__":
    main()
