#!/usr/bin/env python3
"""
中煤鄂尔多斯能源化工有限公司 — 能源与环境栏目爬虫
URL: https://enh.chinacoal.com/col/col2885/index.html
CMS: Hanweb 6.1.7.1 (API动态加载)
API: GET /api-gateway/jpaas-publish-server/front/page/build/unit
Detail: <div id="zoomFont" class="wzy_bd"> → <p> content
"""

import requests
from bs4 import BeautifulSoup
import json
import re
import sys
import os
import argparse
from datetime import datetime

# ─── 配置 ─────────────────────────────────────────────
BASE_URL = "https://enh.chinacoal.com"
LIST_API = "https://enh.chinacoal.com/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_NAME = "中煤鄂尔多斯能源化工有限公司"
COLUMN_NAME = "能源与环境"
GROUP = "企业"
INDUSTRY = "环境公示"
ROWS = 15  # 每页条数

BASE_API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "35",
    "tplSetId": "J8JwwwtafcbqyOlIHQvaF",
    "pageType": "column",
    "tagId": "当前栏目_list",
    "editType": "null",
    "pageId": "2885",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import fetch_page, push_to_searchdb


def fetch_list_page(page_no: int) -> list:
    """获取列表页，返回 [(title, url, pub_date), ...]"""
    params = {**BASE_API_PARAMS}
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": ROWS}, ensure_ascii=False)
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=15)
        resp.raise_for_status()
        data = resp.json()
    except Exception as e:
        print(f"  ❌ 列表页 {page_no} 请求失败: {e}", flush=True)
        return []

    html = data.get("data", {}).get("html", "")
    if not html:
        print(f"  ❌ 列表页 {page_no} 无HTML数据", flush=True)
        return []

    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("ul.lmy_tylb1 li"):
        a_tag = li.find("a")
        span_tag = li.find("span")
        if not a_tag or not a_tag.get("href"):
            continue
        href = a_tag["href"].strip()
        title = a_tag.get("title", "").strip() or a_tag.get_text(strip=True)
        if not title:
            continue
        # 处理相对URL
        if href.startswith("/"):
            url = BASE_URL + href
        else:
            url = href
        date_str = span_tag.get_text(strip=True) if span_tag else ""
        items.append((title, url, date_str))

    print(f"  📄 第{page_no}页: {len(items)}条", flush=True)
    return items


def fetch_detail(url: str) -> tuple:
    """
    获取详情页，返回 (title, content, publish_date)
    返回 (None, ..., ...) 表示失败
    """
    resp = fetch_page(url)
    if resp is None:
        return None, "", ""

    html = resp
    soup = BeautifulSoup(html, "html.parser")

    # 标题: 优先 ArticleTitle meta
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True)

    # 日期
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()

    # 正文: <div id="zoomFont" class="wzy_bd">
    content_div = soup.find("div", id="zoomFont", class_="wzy_bd")
    if not content_div:
        content_div = soup.find("div", id="zoomFont")
    if not content_div:
        content_div = soup.find("div", class_="wzy_bd")

    content = ""
    if content_div:
        # 移除script/style
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()
        # 获取p标签文本
        paragraphs = []
        for p in content_div.find_all("p"):
            text = p.get_text(strip=True)
            if text and text not in paragraphs:
                paragraphs.append(text)
        content = "\n".join(paragraphs) if paragraphs else content_div.get_text(strip=True)

    return title, content, pub_date


def get_total_count() -> int:
    """获取总条数"""
    params = {**BASE_API_PARAMS}
    params["paramJson"] = json.dumps({"pageNo": 1, "pageSize": ROWS}, ensure_ascii=False)
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=15)
        resp.raise_for_status()
        data = resp.json()
        html = data.get("data", {}).get("html", "")
        m = re.search(r'count="(\d+)"', html)
        if m:
            return int(m.group(1))
    except Exception:
        pass
    return 0


def crawl(pages: int = 5, push: bool = True):
    """主爬虫"""
    total = get_total_count()
    total_pages = max(1, (total + ROWS - 1) // ROWS) if total else 5
    actual_pages = min(pages, total_pages) if pages > 0 else total_pages

    print(f"🔍 {SITE_NAME} - {COLUMN_NAME}", flush=True)
    print(f"📊 总计 {total} 条, {total_pages} 页, 本次爬取 {actual_pages} 页", flush=True)

    all_items = []
    for page in range(1, actual_pages + 1):
        items = fetch_list_page(page)
        if not items:
            print(f"  ⚠️ 第{page}页为空，停止翻页", flush=True)
            break
        all_items.extend(items)

    print(f"\n📋 列表共 {len(all_items)} 条", flush=True)
    print(f"{'='*60}", flush=True)

    push_items = []
    fail_count = 0

    for idx, (title, url, date_str) in enumerate(all_items, 1):
        print(f"\n[{idx}/{len(all_items)}] {title[:60]}", flush=True)
        print(f"    📎 {url}", flush=True)
        print(f"    📅 {date_str}", flush=True)

        # 取详情
        detail_title, content, pub_date = fetch_detail(url)
        if detail_title is None:
            print(f"    ❌ 详情获取失败", flush=True)
            fail_count += 1
            continue

        final_title = detail_title or title
        final_date = pub_date or date_str

        content_preview = content[:80].replace('\n', ' ')
        print(f"    ✅ 正文 {len(content)} 字: {content_preview}...", flush=True)

        push_items.append({
            "title": final_title,
            "url": url,
            "source_url": url,
            "pub_date": final_date,
            "content": content,
            "summary": content[:500] if content else final_title,
            "site_name": SITE_NAME,
            "group_name": GROUP,
            "industry": INDUSTRY,
            "category": COLUMN_NAME,
        })

    # 批量推送到search.db
    if push and push_items:
        print(f"\n{'='*60}", flush=True)
        print(f"📤 正在推送 {len(push_items)} 条到 search.db...", flush=True)
        push_to_searchdb(push_items, batch_label=f"{SITE_NAME}-{COLUMN_NAME}")
    elif push:
        print(f"\n⚠️ 无数据可推送", flush=True)

    print(f"\n{'='*60}", flush=True)
    print(f"✅ 完成: 共{len(push_items)}条, 失败{fail_count}", flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=f"{SITE_NAME} - {COLUMN_NAME} 爬虫")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数 (0=全部)")
    parser.add_argument("--no-push", action="store_true", help="不推送到search.db")
    args = parser.parse_args()

    crawl(pages=args.pages, push=not args.no_push)
