#!/usr/bin/env python3
"""
中煤陕西能源化工集团有限公司 — 通知公告
Hanweb CMS, API分页, 单页
"""
import requests
import re
import json
import sys
import os
import argparse
from bs4 import BeautifulSoup

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

SITE_NAME = "中煤陕西能源化工集团有限公司"
DOMAIN = "shaanxi.chinacoal.com"
BASE = "https://shaanxi.chinacoal.com"
API_URL = "https://shaanxi.chinacoal.com/api-gateway/jpaas-publish-server/front/page/build/unit"
GROUP = "企业"
INDUSTRY = "企业环保"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "17",
    "tplSetId": "EMwhExxk3yqkuApq4CDyY",
    "pageType": "column",
    "tagId": "当前栏目_list",
    "editType": "null",
    "pageId": "1470",
}
PAGE_SIZE = 15


def fetch_list():
    """Fetch all items via Hanweb API."""
    param_json = json.dumps({"pageNo": 1, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    params = {**API_PARAMS, "paramJson": param_json}
    r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    if r.status_code != 200:
        return []
    data = r.json()
    if not data.get("success"):
        return []
    html = data["data"]["html"]
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("li"):
        a = li.find("a")
        if a:
            href = a.get("href", "")
            title = a.get("title", a.get_text(strip=True))
            span = li.find("span")
            date = span.get_text(strip=True) if span else ""
            if href:
                full_url = href if href.startswith("http") else BASE + href
                items.append({"url": full_url, "title": title.strip(), "date": date.strip()})
    return items


def fetch_detail(url):
    """Fetch detail page."""
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    if r.status_code != 200:
        return None
    soup = BeautifulSoup(r.text, "html.parser")

    # Title from meta
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()

    # Date from meta
    date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        raw = meta_date["content"].strip()
        m = re.match(r"(\d{4})-(\d{1,2})-(\d{1,2})", raw)
        if m:
            date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"

    # Body from div.wzy_bd.article
    body_text = ""
    content = soup.select_one("div.wzy_bd.article")
    if not content:
        content = soup.select_one("div.wzy_bd")
    if content:
        for s in content.find_all("script"):
            s.decompose()
        parts = []
        for p in content.find_all("p"):
            t = p.get_text(strip=True)
            if t:
                parts.append(t)
        body_text = "\n".join(parts)

    return {"title": title, "date": date, "body_text": body_text}


def main():
    parser = argparse.ArgumentParser(description="中煤陕西通知公告爬虫")
    parser.add_argument("--push", action="store_true", default=False)
    args = parser.parse_args()

    import urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

    print(f"{SITE_NAME} 通知公告 — 爬取", flush=True)

    items = fetch_list()
    if not items:
        print("无数据", file=sys.stderr)
        return
    print(f"列表: {len(items)} 条", flush=True)

    details = []
    for idx, item in enumerate(items, 1):
        print(f"  [{idx}/{len(items)}] {item['title'][:35]}...", end=" ", flush=True)
        detail = fetch_detail(item["url"])
        if not detail:
            print("❌", flush=True)
            continue
        details.append({
            "site_name": SITE_NAME,
            "source_url": item["url"],
            "url": item["url"],
            "title": detail["title"] or item["title"],
            "pub_date": detail["date"],
            "summary": (detail["body_text"] or "")[:500],
            "content": detail["body_text"] or "",
            "category": GROUP,
            "tags": INDUSTRY,
            "attachments": "",
        })
        print(f"✅ {len(detail['body_text'])}字", flush=True)

    saved, skipped = len(details), len(items) - len(details)
    print(f"\n完成: {saved} 条, {skipped} 条失败", flush=True)

    if args.push and details:
        push_to_searchdb(details, batch_label="shaanxi_chinacoal")

    output = {"saved": saved, "skipped": skipped, "total": len(items), "push": args.push}
    print(f"\n---STATS---\n{json.dumps(output)}")


if __name__ == "__main__":
    main()
