#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
来安县人民政府 - 其他权力（环评相关）爬虫
https://www.laian.gov.cn/public/column/161054725?type=4&catId=170002617&action=list
CMS: 滁州市政府信息公开平台 (安徽)
列表: ul.xxgk_navli2 > li (a.title + span.date)
分页: 单页7条，无分页
详情: div.gkwz_contnet.j-fontContent 正文
"""

import requests
import re
import sqlite3
import time
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

LIST_URL = "https://www.laian.gov.cn/public/column/161054725?type=4&catId=170002617&action=list"
BASE_URL = "https://www.laian.gov.cn"
SITE_NAME = "来安县生态环境-其他权力"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

DB_PATH = "/root/search.db"
DAYS_BACK = 365


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.text_factory = str
    return conn


def url_exists(conn, url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (url,))
    return cur.fetchone() is not None


def insert_record(conn, record):
    conn.execute(
        """INSERT OR IGNORE INTO gov_raw
           (title, page_url, summary, content, site_name, publish_date)
           VALUES (?, ?, ?, ?, ?, ?)""",
        (record["title"], record["page_url"], record["summary"],
         record["content"], record["site_name"],
         record["publish_date"])
    )


def fetch_list():
    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code == 200:
            return resp.text
        return None
    except Exception as e:
        print(f"[ERROR] 列表页请求失败: {e}")
        return None


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for ul in soup.find_all("ul", class_="xxgk_navli2"):
        li = ul.find("li")
        if not li:
            continue
        a_tag = li.find("a", class_="title")
        span_date = li.find("span", class_="date")
        if not a_tag:
            continue
        
        href = a_tag.get("href", "").strip()
        title = a_tag.get_text(strip=True)
        pub_date = span_date.get_text(strip=True) if span_date else ""
        
        if not href or not title:
            continue
        
        # 补全URL
        if href.startswith("http"):
            full_url = href
        elif href.startswith("/"):
            full_url = BASE_URL + href
        else:
            full_url = BASE_URL + "/" + href
        
        items.append((title, full_url, pub_date))
    
    return items


def fetch_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, []
        
        soup = BeautifulSoup(resp.text, "html.parser")
        
        # 正文
        content_div = soup.find("div", class_="gkwz_contnet")
        if not content_div:
            content_div = soup.find("div", class_="xxgk_contnet")
        if not content_div:
            # fallback
            for cls in ["contentbox", "xxgk_content", "article-content"]:
                content_div = soup.find("div", class_=cls)
                if content_div:
                    break
        
        content_text = ""
        content_html = ""
        if content_div:
            # 提取纯文本（带段落分隔）
            content_text = content_div.get_text(separator='\n\n').strip()
            # 保留HTML原始内容
            content_html = str(content_div)
        
        # 附件
        attachments = []
        if content_div:
            for a in content_div.find_all("a"):
                href = a.get("href", "").strip()
                text = a.get_text(strip=True)
                if not href or href.startswith("#") or "javascript" in href.lower():
                    continue
                if href.startswith("/"):
                    ahref = BASE_URL + href
                elif not href.startswith("http"):
                    ahref = BASE_URL + "/" + href
                else:
                    ahref = href
                
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
                    attachments.append({"text": text, "url": ahref})
        
        return content_html or content_text, attachments
    
    except Exception as e:
        print(f"[ERROR] 详情页 {url} 请求失败: {e}")
        return None, []


def should_crawl(pub_date):
    if pub_date:
        try:
            pub_dt = datetime.strptime(pub_date, "%Y-%m-%d")
            if pub_dt < datetime.now() - timedelta(days=DAYS_BACK):
                return False
        except:
            pass
    return True


def main():
    print(f"=== 来安县生态环境-其他权力爬虫 ===")
    print(f"目标: {LIST_URL}")
    print(f"时间范围: 最近{DAYS_BACK}天\n")
    
    conn = get_conn()
    
    html = fetch_list()
    if not html:
        print("[ERROR] 无法获取列表页")
        conn.close()
        return
    
    items = parse_list(html)
    print(f"列表共 {len(items)} 条\n")
    
    total_found = len(items)
    total_inserted = 0
    total_skipped = 0
    total_outdated = 0
    
    for title, url, pub_date in items:
        if not should_crawl(pub_date):
            print(f"  [-] {title[:50]}... ({pub_date}) 超出时间范围")
            total_outdated += 1
            continue
        
        if url_exists(conn, url):
            print(f"  [-] {title[:50]}... 已存在")
            total_skipped += 1
            continue
        
        content, attachments = fetch_detail(url)
        if not content or len(content) < 20:
            print(f"  [!] {title[:50]}... 内容为空")
            total_skipped += 1
            continue
        
        # summary用纯文本（去HTML标签）
        summary_plain = re.sub(r'<[^>]+>', '', content)[:200] if content else ''
        summary = summary_plain
        
        record = {
            "title": title,
            "page_url": url,
            "summary": summary,
            "content": content,
            "site_name": SITE_NAME,
            "publish_date": pub_date,
        }
        
        insert_record(conn, record)
        conn.commit()
        total_inserted += 1
        print(f"  [+] {title[:50]}... ({pub_date})")
        if attachments:
            print(f"     附件: {len(attachments)}个")
        
        time.sleep(0.5)
    
    conn.close()
    
    print(f"\n=== 完成 ===")
    print(f"总扫描: {total_found} 条")
    print(f"新增入库: {total_inserted} 条")
    print(f"已存在: {total_skipped} 条")
    print(f"超出时间范围: {total_outdated} 条")


if __name__ == "__main__":
    main()
