#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
利华益集团 - 安环公开栏目爬虫
WebsiteOnline CMS
https://www.lihuayi.com/news?article_category=10&brd=1

列表页: HTML rendered (server-side), ?page=N 分页
详情页: /page95?article_id=NNNN
"""

import sys
import os
import re
import requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from crawler_lib import push_to_searchdb, clean_html

# ===== 配置 =====
BASE_URL = "https://www.lihuayi.com"
LIST_URL = "https://www.lihuayi.com/news?article_category=10&brd=1"
SITE_NAME = "利华益集团"
MAX_PAGES = 5  # 前5页
INDUSTRY = "环境公示"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def fetch_page(page=1):
    """获取列表页HTML"""
    url = f"{LIST_URL}&page={page}"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"[错误] 列表页{page}请求失败: {e}")
        return ""


def parse_list(html):
    """解析列表页，提取文章信息"""
    soup = BeautifulSoup(html, "html.parser")
    items = []

    # 找到所有文章li
    articles = soup.select("li.wpart-border-line")
    for li in articles:
        try:
            # 标题和URL
            title_link = li.select_one("p.title a.articleid")
            if not title_link:
                continue
            title = title_link.get("title", "").strip()
            url = title_link.get("href", "").strip()

            if not title or not url:
                continue

            # 日期
            time_tag = li.select_one("p.time .wp-new-ar-pro-time")
            pub_date = time_tag.get_text(strip=True) if time_tag else ""

            # 摘要
            abstract_tag = li.select_one("p.abstract")
            summary = ""
            if abstract_tag:
                text = abstract_tag.get_text(strip=True)
                if text.startswith("摘要："):
                    text = text[3:]
                summary = text.strip()

            # 分类
            cat_tag = li.select_one("p.title span.category")
            category = cat_tag.get_text(strip=True) if cat_tag else ""

            items.append({
                "title": title,
                "url": url if url.startswith("http") else BASE_URL + url,
                "pub_date": pub_date,
                "summary": summary,
                "category": category,
            })
        except Exception as e:
            print(f"  [解析错误] {e}")
            continue

    return items


def fetch_detail(url):
    """获取详情页内容"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")

        # 正文
        detail_div = soup.select_one(".artview_detail")
        content_html = str(detail_div) if detail_div else ""

        # 清理HTML保留表格
        content = clean_html(content_html)

        return content
    except Exception as e:
        print(f"  [详情错误] {url}: {e}")
        return ""


def main():
    all_items = []
    seen_urls = set()

    for page in range(1, MAX_PAGES + 1):
        print(f"[列表] 页{page}...")
        html = fetch_page(page)
        if not html:
            break

        items = parse_list(html)
        print(f"  -> 解析 {len(items)} 条")

        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])

            # 取详情页内容
            content = fetch_detail(item["url"])
            if content:
                print(f"  [OK] {item['title'][:30]} | {item['pub_date']}")
            else:
                print(f"  [无正文] {item['title'][:30]} | {item['pub_date']}")

            all_items.append({
                "title": item["title"],
                "url": item["url"],
                "source_url": item["url"],
                "site_name": SITE_NAME,
                "pub_date": item["pub_date"],
                "summary": item["summary"],
                "content": content,
                "industry": INDUSTRY,
                "category": item.get("category", ""),
            })

    print(f"\n[汇总] 共 {len(all_items)} 条待入库")

    # 分批入库
    batch_size = 20
    for i in range(0, len(all_items), batch_size):
        batch = all_items[i:i + batch_size]
        try:
            result = push_to_searchdb(batch, batch_label="lihuayi")
            print(f"  [批 {i//batch_size + 1}] 入库 {len(batch)} 条")
        except Exception as e:
            print(f"  [批 {i//batch_size + 1}] 错误: {e}")

    print(f"\n===== 完成 =====")
    print(f"总计: {len(all_items)} 条")


if __name__ == "__main__":
    main()
