#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
辽宁昌鑫环境工程咨询有限公司 - 公示信息爬虫
中企动力 (300.cn) CMS
http://www.liaoningchangxin.com/news/7/  (公示信息栏目, CID: 1690997705924546560)

列表页: POST /fwebapi/cms/lowcode/60003/18505/list?cate=0
详情页: /news_details/{id}.html 或 /news/{id}.html
"""

import sys
import os
import re
import json
import requests
from bs4 import BeautifulSoup
from datetime import datetime

# 确保能找到上级目录的库
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from crawler_lib import push_to_searchdb, clean_html

# ===== 配置 =====
BASE_URL = "http://www.liaoningchangxin.com"
LIST_API = "http://www.liaoningchangxin.com/fwebapi/cms/lowcode/60003/18505/list?cate=0"
CID = "1690997705924546560"  # 公示信息栏目
TOTAL_PAGES = 3  # 586条, 每页200条, 共3页
PAGE_SIZE = 200
SITE_NAME = "辽宁昌鑫环境工程咨询有限公司"
INDUSTRY = "环评公示"  # 企业环保类

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Content-Type": "application/json",
    "Referer": "http://www.liaoningchangxin.com/news/7/",
}

session = requests.Session()


def fetch_list(page=0):
    """获取列表页数据"""
    payload = {
        "size": PAGE_SIZE,
        "query": [
            {
                "valueName": "",
                "dataType": "array[category]",
                "operator": "in",
                "filter": "ignore-empty-check",
                "esField": "DETAIL_ES.es_multi_category_6d5k7017",
                "groupName": "数据展示条件,默认条件组",
                "groupEnd": "2,1",
                "field": "category_6d5k7017",
                "sourceType": "static",
                "logic": "and",
                "groupBegin": "1,2",
                "value": CID,
                "fieldType": "array",
            }
        ],
        "header": {
            "Data-Query-Es-Field": "DETAIL_ES.es_symbol_text_2PVENH84,TEXT_DETAIL_ES.es_text_textarea_8B30N7U4,DETAIL_ES.es_date_prePublishTime,DETAIL_ES.es_date_preFirstPublishTime",
            "Data-Query-Random": 0,
            "Data-Query-Field": "text_2PVENH84,textarea_8B30N7U4,prePublishTime,preFirstPublishTime",
        },
        "from": page * PAGE_SIZE,
        "sort": [],
        "_detailId": CID,
    }

    try:
        resp = session.post(LIST_API, json=payload, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        data = resp.json()
        items = data.get("data", {}).get("list", [])
        total = data.get("data", {}).get("page", {}).get("totalCount", 0)
        print(f"[列表] 页{page}, 返回{len(items)}条, 总计{total}条")
        return items
    except Exception as e:
        print(f"[错误] 列表请求失败: {e}")
        return []


def fetch_detail(url):
    """获取详情页内容"""
    full_url = BASE_URL + url if url.startswith("/") else url
    try:
        resp = session.get(full_url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
        soup = BeautifulSoup(html, "html.parser")

        # 提取标题
        title_tag = soup.select_one(".e_h1-20.s_subtitle")
        title = title_tag.get_text(strip=True) if title_tag else ""

        # 提取正文
        content_div = soup.select_one(".e_richText-24.s_title")
        content_html = str(content_div) if content_div else ""

        # 清理HTML保留表格
        content = clean_html(content_html)

        return title, content
    except Exception as e:
        print(f"[错误] 详情页请求失败 {url}: {e}")
        return "", ""


def main():
    all_items = []
    processed_urls = set()

    # 第一步：收集所有列表数据（不取详情，避免重复请求耗费时间）
    for page in range(TOTAL_PAGES):
        items = fetch_list(page)
        if not items:
            break

        for item in items:
            title = item.get("text_2PVENH84", "").strip()
            detail_url = item.get("_href", "")
            pub_time = item.get("prePublishTime", "")
            summary = item.get("textarea_8B30N7U4", "") or ""
            content = ""  # 后续通过详情页补充
            
            if not title or not detail_url:
                continue
            
            # 先去重
            if detail_url in processed_urls:
                continue
            processed_urls.add(detail_url)

            # 格式化日期
            publish_date = ""
            if pub_time:
                try:
                    dt = datetime.strptime(pub_time[:10], "%Y-%m-%d")
                    publish_date = dt.strftime("%Y-%m-%d")
                except:
                    pass

            source_url = BASE_URL + detail_url if detail_url.startswith("/") else detail_url

            all_items.append({
                "title": title,
                "url": source_url,
                "source_url": source_url,
                "site_name": SITE_NAME,
                "pub_date": publish_date,
                "summary": summary,
                "content": content,
                "industry": INDUSTRY,
            })

    print(f"\n[汇总] 共 {len(all_items)} 条待入库")

    # 第二步：分批入库
    batch_size = 50
    for i in range(0, len(all_items), batch_size):
        batch = all_items[i:i + batch_size]
        try:
            result = push_to_searchdb(batch, batch_label="liaoningchangxin")
            print(f"  [批 {i//batch_size + 1}] 入库 {len(batch)} 条 -> {result}")
        except Exception as e:
            print(f"  [批 {i//batch_size + 1}] 错误: {e}")

    print(f"\n===== 完成 =====")
    print(f"总计: {len(all_items)} 条")


if __name__ == "__main__":
    main()
