#!/usr/bin/env python3
"""寻甸县-生态环境 爬虫"""
import requests, re, os
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "寻甸县-生态环境"
BASE_URL = "http://www.kmxd.gov.cn"
LIST_PATH = "/zfxxgkml/fdzdgknr/shgysyjsly/hjbh"
TOTAL_PAGES = 25

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ❌ {e}")
        return None


def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for div in soup.find_all("div", class_="data-table-item"):
        a = div.find("a", href=True)
        if not a: continue
        href = a.get("href", "")
        title = a.get_text(strip=True)
        if not href or not title: continue
        detail_url = href if href.startswith("http") else BASE_URL + href
        # Date from nearby element
        date_str = ""
        date_el = div.find("p", class_="w80")
        if date_el:
            date_a = date_el.find("a")
            if date_a:
                date_str = date_a.get_text(strip=True).replace(".", "-")
        items.append((title, detail_url, date_str))
    return items


def parse_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"]
    publish_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r'(\d{4}-\d{2}-\d{2})', meta_date["content"])
        if m: publish_date = m.group(1)
    content = ""
    content_div = soup.find("div", class_="content")
    if content_div:
        content = str(content_div)
    return title, content, publish_date


def crawl():
    conn = None
    try:
        import sqlite3
        conn = sqlite3.connect(SEARCH_DB)
        c = conn.cursor()
        c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT, content TEXT,
            source_url TEXT UNIQUE, site_name TEXT, publish_date TEXT)''')
        total_new = total_skip = 0

        for page in range(1, TOTAL_PAGES + 1):
            if page == 1:
                page_url = BASE_URL + LIST_PATH + "/index.shtml"
            else:
                page_url = BASE_URL + LIST_PATH + f"/index_{page}.shtml"
            html = fetch(page_url)
            if not html: continue
            items = parse_list(html)
            if not items:
                print(f"第{page}/{TOTAL_PAGES}页: 无数据，结束")
                break
            print(f"第{page}/{TOTAL_PAGES}页: {len(items)}条")

            for i, (title, detail_url, date_from_list) in enumerate(items, 1):
                c.execute("SELECT id FROM gov_raw WHERE source_url = ?", (detail_url,))
                if c.fetchone():
                    total_skip += 1
                    continue
                detail_html = fetch(detail_url)
                if not detail_html:
                    print(f"  [{i}] ❌ 详情失败: {title[:30]}...")
                    continue
                _, content, publish_date = parse_detail(detail_html)
                if not publish_date: publish_date = date_from_list
                try:
                    c.execute(
                        "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                        (title, content, detail_url, SITE_NAME, publish_date or "")
                    )
                    conn.commit()
                    total_new += 1
                    content_len = len(content) if content else 0
                    status = " ⚠️ 过短" if content_len < 200 else ""
                    if total_new <= 5 or content_len < 200:
                        print(f"  [{i}] ✅ {title[:35]}... ({publish_date}){status}")
                except Exception as e:
                    print(f"  [{i}] ❌ 入库失败: {e}")

        print(f"\n✅ 新增: {total_new}, 跳过: {total_skip}")
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        print(f"  共 {c.fetchone()[0]} 条")
    finally:
        if conn: conn.close()

if __name__ == "__main__":
    crawl()
