#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
连山区人民政府 - 项目环评
TRS CMS
http://www.lianshan.gov.cn/zwgk/zwgkzdgz/shgysy/hjbh/xmhp/

列表页: ul.news_ul > li (title + date)
分页: index_N.html (21条/页)
详情: div.TRS_Editor
"""

import sys
import os
import re
import argparse
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from crawler_lib import push_to_searchdb, clean_html

# ===== 配置 =====
BASE_URL = "http://www.lianshan.gov.cn/zwgk/zwgkzdgz/shgysy/hjbh/xmhp/"
SITE_NAME = "连山区人民政府-项目环评"
INDUSTRY = "环评公示"
PAGE_SIZE = 21

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}


def clean_title(title):
    """清理标题中的 &middot; &nbsp; 等实体前缀"""
    title = re.sub(r'^\s*(&middot;|&nbsp;|\s|·|\s)+', '', title)
    title = re.sub(r'(&middot;|&nbsp;)+', '', title)
    return title.strip()


def fetch_list(page=1):
    """获取列表页HTML"""
    if page == 1:
        url = BASE_URL
    else:
        url = f"{BASE_URL}index_{page}.html"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"[错误] 列表页{page}: {e}")
        return ""


def parse_list(html):
    """解析列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []

    for li in soup.select("ul.news_ul > li"):
        try:
            link = li.select_one("a")
            if not link:
                continue

            title = link.get("title", "") or link.get_text(strip=True) or ""
            title = clean_title(title)
            href = link.get("href", "").strip()

            if not title or not href:
                continue

            # 相对URL转绝对
            full_url = urljoin(BASE_URL, href)

            # 日期
            date_span = li.select_one("span")
            pub_date = ""
            if date_span:
                d = date_span.get_text(strip=True).strip("[]").strip()
                pub_date = d

            items.append({
                "title": title,
                "url": full_url,
                "pub_date": pub_date,
            })
        except Exception as e:
            print(f"  [解析错误] {e}")
            continue

    return items


def fetch_detail(url):
    """获取详情页正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")

        # 正文在 div.TRS_Editor
        editor = soup.select_one("div.TRS_Editor")
        content_html = str(editor) if editor else ""

        # 清理
        content = clean_html(content_html)

        return content
    except Exception as e:
        print(f"  [详情错误] {url}: {e}")
        return ""


def main():
    parser = argparse.ArgumentParser(description="连山区-项目环评爬虫")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数")
    args = parser.parse_args()

    max_pages = args.pages
    all_items = []
    seen_urls = set()

    for page in range(1, max_pages + 1):
        print(f"[列表] 页{page}...")
        html = fetch_list(page)
        if not html:
            print("  -> 无响应")
            break

        items = parse_list(html)
        if not items:
            print("  -> 无数据，结束")
            break

        print(f"  -> {len(items)} 条")

        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])

            # 取详情
            content = fetch_detail(item["url"])
            status = "OK" if content else "空"
            print(f"  [{status}] {item['title'][:40]} | {item['pub_date']}")

            all_items.append({
                "title": item["title"],
                "url": item["url"],
                "source_url": item["url"],
                "site_name": SITE_NAME,
                "pub_date": item["pub_date"],
                "summary": "",
                "content": content,
                "industry": INDUSTRY,
            })

    print(f"\n[汇总] 共 {len(all_items)} 条")

    if all_items:
        # 分批入库
        batch_size = 30
        for i in range(0, len(all_items), batch_size):
            batch = all_items[i:i + batch_size]
            try:
                push_to_searchdb(batch, batch_label="lianshan_xmhp")
                print(f"  [批 {i//batch_size + 1}] 入库 {len(batch)} 条")
            except Exception as e:
                print(f"  [批 {i//batch_size + 1}] 错误: {e}")

    print(f"\n===== 完成 =====")
    print(f"总计: {len(all_items)} 条")


if __name__ == "__main__":
    main()
