#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
连山区人民政府 - 通知公告
CMS: TRS
列表: https://www.lianshan.gov.cn/lssx/tzgg/index.html (index_2.html ...)
详情: div.TRS_Editor
"""

import requests
import sqlite3
import re
import time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.lianshan.gov.cn/lssx/tzgg/"
DB_PATH = "/root/search.db"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 5
SITE_NAME = "连山区通知公告"
DELAY = 0.5

def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] Fetch failed: {url} - {e}")
        return None

def parse_list_page(soup):
    items = []
    ul = soup.find("ul", class_="news_ul")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a.get("href", "").strip()
        if not href or "javascript" in href:
            continue
        full_url = urljoin(BASE_URL, href)
        title = a.get("title") or a.get_text(strip=True)
        if not title:
            continue
        date_str = ""
        span = li.find("span")
        if span:
            date_str = span.get_text(strip=True).strip("[] ")
        items.append((title, full_url, date_str))
    return items

def parse_detail(soup):
    content = ""
    # TRS_Editor with or without quotes
    editor = soup.find("div", class_="TRS_Editor")
    if editor:
        content = str(editor)
    if not content:
        editor = soup.find("div", attrs={"class": "TRS_Editor"})
        if editor:
            content = str(editor)
    if not content:
        maintxt = soup.find("div", id="maintxt")
        if maintxt:
            content = str(maintxt)
    
    title = ""
    p = soup.find("p", class_="title_news")
    if p:
        title = p.get_text(strip=True)
    
    date_str = ""
    source = ""
    msg = soup.find("div", class_="title_message")
    if msg:
        texts = msg.get_text(" ", strip=True)
        m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", texts)
        if m:
            date_str = m.group(1)
    
    return title, content, date_str, source

def main():
    print(f"[{SITE_NAME}] 开始爬取...")
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, MAX_PAGES + 1):
        url = urljoin(BASE_URL, f"index_{page}.html") if page > 1 else BASE_URL + "index.html"
        print(f"\n--- 第{page}页: {url}")
        soup = get_soup(url)
        if not soup:
            total_error += 1
            continue
        items = parse_list_page(soup)
        print(f"  列表项: {len(items)}")
        if not items:
            print("  无更多列表项，停止")
            break

        for title, page_url, list_date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue
            time.sleep(DELAY)
            detail_soup = get_soup(page_url)
            if not detail_soup:
                total_error += 1
                continue
            detail_title, content, pub_date, source = parse_detail(detail_soup)
            text_len = len(re.sub(r'<[^>]+>', '', content).strip())
            if text_len < 50:
                print(f"  跳过(正文太短{text_len}): {title[:40]}")
                total_skip += 1
                continue
            final_title = detail_title or title
            final_date = pub_date or list_date
            clean_text = re.sub(r'<[^>]+>', '', content).strip()
            summary = clean_text[:200]
            try:
                c.execute("""INSERT INTO gov_raw (site_name, title, content, page_url, source_url, publish_date, summary, status, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?, 'active', 1)""",
                          (SITE_NAME, final_title, content, page_url, "", final_date, summary))
                conn.commit()
                total_new += 1
                print(f"  +1 [{final_date}] {final_title[:50]}")
            except Exception as e:
                print(f"  [ERROR] DB insert: {e}")
                conn.rollback()
    conn.close()
    print(f"\n=== 完成 ===\n新增: {total_new}, 跳过: {total_skip}, 错误: {total_error}")

if __name__ == "__main__":
    main()
