#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
荔波县人民政府 - 环境保护栏目爬虫
贵州政府网站群 (TRS CMS)
http://www.libo.gov.cn/zwgk/xxgkml/zdlyxx/hjbh/

列表页: HTML table解析，index_N.html分页
详情页: 正文在 div.article
"""

import sys
import os
import re
import requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from crawler_lib import push_to_searchdb, clean_html

# ===== 配置 =====
BASE_URL = "http://www.libo.gov.cn"
LIST_URL = "http://www.libo.gov.cn/zwgk/xxgkml/zdlyxx/hjbh/"
SITE_NAME = "荔波县人民政府"
INDUSTRY = "环境公示"
MAX_PAGES = 40  # 39页满+1页部分

def parse_max_pages():
    """页数: 裸数字 / --pages=N / --pages N (默认40全量)"""
    import sys as _sys
    mp = MAX_PAGES
    for i, a in enumerate(_sys.argv):
        if a.isdigit():
            mp = int(a)
        elif a.startswith("--pages="):
            try:
                mp = int(a.split("=")[1])
            except ValueError:
                pass
        elif a == "--pages" and i + 1 < len(_sys.argv) and _sys.argv[i+1].isdigit():
            mp = int(_sys.argv[i+1])
    return mp

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def fetch_list(page=1):
    """获取列表页HTML"""
    if page == 1:
        url = LIST_URL
    else:
        url = f"{BASE_URL}/zwgk/xxgkml/zdlyxx/hjbh/index_{page}.html"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"[错误] 列表页{page}: {e}")
        return ""


def parse_list(html):
    """解析表格列表"""
    soup = BeautifulSoup(html, "html.parser")
    items = []

    # 找到表格中的每一行
    rows = soup.select("table.bg_tb tbody#idData tr.c")
    for row in rows:
        try:
            # 标题和URL
            link_td = row.select_one("td.tn4 a")
            if not link_td:
                continue
            title = link_td.get_text(strip=True)
            href = link_td.get("href", "").strip()
            if not title or not href:
                continue
            if not href.startswith("http"):
                href = BASE_URL + href

            # 日期
            date_td = row.select_one("td.tn5")
            pub_date = date_td.get_text(strip=True) if date_td else ""

            # 发布机构
            body_td = row.select_one("td.tn2")
            body = body_td.get_text(strip=True) if body_td else ""

            items.append({
                "title": title,
                "url": href,
                "pub_date": pub_date,
                "body": body,
            })
        except Exception as e:
            print(f"  [解析错误] {e}")
            continue

    return items


def fetch_detail(url):
    """获取详情页正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")

        # 正文在 div.Article_zw（实际内容区，不含元数据）
        article = soup.select_one("div.Article_zw")
        content_html = str(article) if article else ""

        # 清理HTML保留表格
        content = clean_html(content_html)

        # 如果正文为空，尝试从页面的纯文本提取
        if not content or len(content) < 50:
            # 提取所有可见文本
            text = soup.get_text(separator="\n")
            lines = [l.strip() for l in text.split("\n") if l.strip() and len(l.strip()) > 10]
            content = "\n".join(lines[10:])  # 跳过头部

        return content
    except Exception as e:
        print(f"  [详情错误] {url}: {e}")
        return ""


def main():
    all_items = []
    seen_urls = set()

    mp = parse_max_pages()
    print(f"[{SITE_NAME}] max_pages={mp}", flush=True)
    for page in range(1, mp + 1):
        print(f"[列表] 页{page}...")
        html = fetch_list(page)
        if not html:
            break

        items = parse_list(html)
        if not items:
            print(f"  -> 无数据，结束")
            break

        print(f"  -> {len(items)} 条")

        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])

            # 取详情页内容
            content = fetch_detail(item["url"])
            status = "OK" if content else "无正文"
            print(f"  [{status}] {item['title'][:40]} | {item['pub_date']}")

            all_items.append({
                "title": item["title"],
                "url": item["url"],
                "source_url": item["url"],
                "site_name": SITE_NAME,
                "pub_date": item["pub_date"],
                "summary": "",
                "content": content,
                "industry": INDUSTRY,
            })

    print(f"\n[汇总] 共 {len(all_items)} 条")

    # 分批入库
    batch_size = 30
    for i in range(0, len(all_items), batch_size):
        batch = all_items[i:i + batch_size]
        try:
            push_to_searchdb(batch, batch_label="libo")
            print(f"  [批 {i//batch_size + 1}] 入库 {len(batch)} 条")
        except Exception as e:
            print(f"  [批 {i//batch_size + 1}] 错误: {e}")

    print(f"\n===== 完成 =====")
    print(f"总计: {len(all_items)} 条")


if __name__ == "__main__":
    main()
