#!/usr/bin/env python3
"""调兵山市人民政府 - 公示公告 爬虫
https://www.lndbss.gov.cn/diaobingshan/xwzx/gsgg/index.html
易思特CMS，10条/页，35页343条
"""

import re
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup
import sqlite3
import os

BASE_URL = "https://www.lndbss.gov.cn"
LIST_PATH = "/diaobingshan/xwzx/gsgg"
SITE_NAME = "lndbss.gov.cn-公示公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATEGORY = "eia"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
CUTOFF_DATE = "2023-06-16"
PAGE_SIZE = 10


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.row_factory = sqlite3.Row
    return conn


def clean_text(text):
    if not text:
        return ""
    return re.sub(r'\s+', ' ', text.strip())


def get_meta(html, name):
    m = re.search(rf'<meta[^>]*name=["\']{name}["\'][^>]*content=["\']([^"\']+)["\']', html)
    return m.group(1) if m else ''


def extract_detail(url):
    """获取详情页内容"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [WARN] 获取详情页失败 {url}: {e}", flush=True)
        return None, None, None, None

    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')

    # 标题
    title = get_meta(html, 'ArticleTitle')
    if not title:
        h2 = soup.select_one('div.article-tit h2')
        if h2:
            title = h2.get_text(strip=True)
    if not title:
        return None, None, None, None
    title = clean_text(title)

    # 日期
    pub_date = get_meta(html, 'PubDate')
    m = re.match(r'(\d{4}-\d{2}-\d{2})', pub_date)
    pub_date = m.group(1) if m else None
    if not pub_date:
        return None, None, None, None

    # 来源
    source = get_meta(html, 'ContentSource')

    # 正文
    content_el = soup.select_one('div.content-box')
    content = str(content_el) if content_el else ''

    # 摘要
    summary = ''
    if content_el:
        text = content_el.get_text(separator=' ', strip=True)
        summary = clean_text(text[:200])

    return title, pub_date, content, summary


def fetch_list_page(page_url):
    """获取列表页，返回文章条目列表和总记录数"""
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [ERROR] 获取列表页失败 {page_url}: {e}", flush=True)
        return [], 0

    html = resp.text

    # 总记录数
    total_m = re.search(r'共(\d+)条', html)
    total_records = int(total_m.group(1)) if total_m else 0

    # 解析文章列表
    items = re.findall(
        r'<li class="clearfix"><a href="([^"]+)"[^>]*>(.*?)</a><span>([^<]+)</span>',
        html
    )

    results = []
    for href, title, date_str in items:
        title = clean_text(title)
        date_str = date_str.strip()
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)
        results.append({
            'title': title,
            'url': href,
            'date': date_str,
            'is_pdf': '.pdf' in href.lower(),
        })

    return results, total_records


def main():
    print(f"=== {SITE_NAME} 爬虫 ===", flush=True)
    print(f"数据库: {DB_PATH}", flush=True)

    conn = get_conn()
    cur = conn.cursor()

    cutoff_dt = datetime.strptime(CUTOFF_DATE, "%Y-%m-%d").date()

    total_new = 0
    total_skip = 0
    total_error = 0

    # 获取第一页判断总页数
    first_articles, total_records = fetch_list_page(BASE_URL + LIST_PATH + "/index.html")
    total_pages = (total_records + PAGE_SIZE - 1) // PAGE_SIZE if total_records else 1
    print(f"总记录: {total_records}, 每页: {PAGE_SIZE}, 总页数: {total_pages}", flush=True)

    all_articles = list(first_articles)

    # 获取剩余页
    for page in range(2, total_pages + 1):
        page_url = BASE_URL + LIST_PATH + f"/0115674e-{page}.html"
        print(f"\n--- 第 {page}/{total_pages} 页: {page_url} ---", flush=True)
        articles, _ = fetch_list_page(page_url)
        if not articles:
            print("  无内容，停止分页", flush=True)
            break
        all_articles.extend(articles)

    print(f"\n共获取 {len(all_articles)} 条列表条目", flush=True)

    for art in all_articles:
        url = art['url']
        list_date = art['date']

        # 检查是否已存在
        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if cur.fetchone():
            total_skip += 1
            continue

        # PDF文章特殊处理（无详情页）
        if art['is_pdf']:
            pub_dt = datetime.strptime(list_date, "%Y-%m-%d").date()
            if pub_dt < cutoff_dt:
                total_skip += 1
                continue

            title = art['title']
            content = f'<div class="pdf-link"><p><a href="{url}" target="_blank">{title}</a></p></div>'
            summary = f"PDF附件：{title}"
            date_rank = int(list_date.replace('-', ''))
            source_url = BASE_URL + LIST_PATH + "/index.html"

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
                    (url, title, content, SITE_NAME, list_date, summary, CATEGORY, date_rank, source_url)
                )
                conn.commit()
                total_new += 1
                print(f"  [PDF] {title[:40]}... | {list_date}", flush=True)
            except Exception as e:
                print(f"  [ERROR] PDF入库失败: {title[:30]}... - {e}", flush=True)
                conn.rollback()
                total_error += 1
            continue

        # HTML文章
        title, pub_date, content, summary = extract_detail(url)

        if not pub_date:
            print(f"  [SKIP] 无详情: {art['title'][:40]}...", flush=True)
            total_error += 1
            continue

        pub_dt = datetime.strptime(pub_date, "%Y-%m-%d").date()
        if pub_dt < cutoff_dt:
            total_skip += 1
            continue

        if not title:
            title = art['title']

        date_rank = int(pub_date.replace('-', ''))
        source_url = BASE_URL + LIST_PATH + "/index.html"

        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
                (url, title, content, SITE_NAME, pub_date, summary, CATEGORY, date_rank, source_url)
            )
            conn.commit()
            total_new += 1
            print(f"  [OK] {title[:40]}... | {pub_date}", flush=True)
        except Exception as e:
            print(f"  [ERROR] 入库失败: {title[:30]}... - {e}", flush=True)
            conn.rollback()
            total_error += 1

    conn.close()
    print(f"\n=== 完成: 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error} ===", flush=True)


if __name__ == '__main__':
    main()
