#!/usr/bin/env python3
"""深圳中环博宏环境技术有限公司 - 信息公告 爬虫
https://www.sz-zhonghuan.com/news/notice.html
DCloud SaaS 平台，共5页67条
"""

import re
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup
import sqlite3
import os

BASE_URL = "https://www.sz-zhonghuan.com"
LIST_URL = urljoin(BASE_URL, "/news/notice.html")
SITE_NAME = "sz-zhonghuan.com-信息公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATEGORY = "eia"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
MAX_PAGES = 5
CUTOFF_DATE = "2023-06-16"


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.row_factory = sqlite3.Row
    return conn


def clean_text(text):
    if not text:
        return ""
    return re.sub(r'\s+', ' ', text.strip())


def extract_full_date(detail_soup):
    """从详情页提取完整日期 (yyyy-MM-dd)"""
    date_el = detail_soup.select_one('p.e_timeFormat-4.s_title')
    if date_el:
        text = date_el.get_text(strip=True)
        m = re.match(r'(\d{4}-\d{2}-\d{2})', text)
        if m:
            return m.group(1)
    return None


def extract_detail(url):
    """获取详情页内容"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [WARN] 获取详情页失败 {url}: {e}", flush=True)
        return None, None, None, None

    soup = BeautifulSoup(resp.text, 'html.parser')

    # 标题
    title_el = soup.select_one('p.e_text-2.s_title')
    title = clean_text(title_el.get_text() if title_el else '')

    # 发布日期
    pub_date = extract_full_date(soup)
    if not pub_date:
        print(f"  [WARN] 详情页无日期: {url}", flush=True)
        return None, None, None, None

    # 正文 (HTML)
    content_el = soup.select_one('div.e_richText-3.s_title.clearfix')
    content = str(content_el) if content_el else ''

    # 摘要 (纯文本前200字)
    summary = ''
    if content_el:
        text = content_el.get_text(separator=' ', strip=True)
        summary = clean_text(text[:200])

    return title, pub_date, content, summary


def fetch_list_page(page_url):
    """获取列表页，返回文章条目列表"""
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [ERROR] 获取列表页失败 {page_url}: {e}", flush=True)
        return []

    soup = BeautifulSoup(resp.text, 'html.parser')
    items = soup.select('div.cbox-3.p_loopitem')

    results = []
    for item in items:
        link_el = item.select_one('p.e_text-7.s_title a')
        if not link_el:
            continue
        href = link_el.get('href', '')
        if not href:
            continue
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)

        title = clean_text(link_el.get_text())

        results.append({
            'title': title,
            'url': href,
        })

    print(f"  列表页解析到 {len(results)} 条", flush=True)
    return results


def main():
    print(f"=== {SITE_NAME} 爬虫 ===", flush=True)
    print(f"数据库: {DB_PATH}", flush=True)

    conn = get_conn()
    cur = conn.cursor()

    cutoff_dt = datetime.strptime(CUTOFF_DATE, "%Y-%m-%d").date()

    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            page_url = LIST_URL
        else:
            page_url = urljoin(BASE_URL, f"/news/notice-{page}.html")

        print(f"\n--- 第 {page}/{MAX_PAGES} 页: {page_url} ---", flush=True)
        articles = fetch_list_page(page_url)

        if not articles:
            print("  无内容，停止分页", flush=True)
            break

        for art in articles:
            url = art['url']

            # 检查是否已存在
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skip += 1
                continue

            title, pub_date, content, summary = extract_detail(url)

            if not pub_date:
                print(f"  [SKIP] 无日期: {art['title'][:40]}...", flush=True)
                total_error += 1
                continue

            pub_dt = datetime.strptime(pub_date, "%Y-%m-%d").date()
            if pub_dt < cutoff_dt:
                total_skip += 1
                continue

            if not title:
                title = art['title']

            # 计算date_rank: YYYYMMDD
            date_rank = int(pub_date.replace('-', ''))

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                    (url, title, content, SITE_NAME, pub_date, summary, CATEGORY, date_rank)
                )
                conn.commit()
                total_new += 1
                print(f"  [OK] {title[:40]}... | {pub_date}", flush=True)
            except Exception as e:
                print(f"  [ERROR] 入库失败: {title[:30]}... - {e}", flush=True)
                conn.rollback()
                total_error += 1

    conn.close()
    print(f"\n=== 完成: 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error} ===", flush=True)


if __name__ == '__main__':
    main()
