#!/usr/bin/env python3
"""
安徽（淮北）新型煤化工合成材料基地管理委员会 — 通知公告
Lonsun CMS, CloudWAF需--curves secp3841r
"""
import subprocess, json, re, sys, os, argparse
from bs4 import BeautifulSoup

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

SITE_NAME = "安徽（淮北）新型煤化工合成材料基地管理委员会"
DOMAIN = "hbmhg.huaibei.gov.cn"
BASE = "https://hbmhg.huaibei.gov.cn"
LIST_URL_P1 = "/xxfb/tzgg/index.html"
LIST_URL_PN = "/content/column/4697996?pageIndex={}"
GROUP = "安徽省"
INDUSTRY = "政府公告"
MAX_PAGES = 50

CURL_BASE = ["curl", "-sL", "--max-time", "15", "--resolve", f"{DOMAIN}:443:103.213.97.228",
             "--curves", "secp384r1", "-H", "User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"]


def curl_get(url):
    result = subprocess.run(CURL_BASE + [url], capture_output=True, timeout=30)
    return result.stdout.decode("utf-8", errors="replace")


def fetch_list(page_no):
    if page_no == 0:
        url = BASE + LIST_URL_P1
    else:
        url = BASE + LIST_URL_PN.format(page_no + 1)
    html = curl_get(url)
    if not html:
        return []
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("ul[class*='doc_list'] li.odd, ul[class*='doc_list'] li.even"):
        a = li.find("a")
        if a and a.get("title"):
            href = a.get("href", "")
            title = a.get("title", "").strip()
            date = ""
            for span in li.find_all("span"):
                txt = span.get_text(strip=True)
                if re.match(r"\d{4}-\d{2}-\d{2}", txt):
                    date = txt
                    break
            if href and title:
                full_url = href if href.startswith("http") else BASE + href
                items.append({"url": full_url, "title": title, "date": date})
    return items


def fetch_detail(url):
    html = curl_get(url)
    if not html:
        return None
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    h1 = soup.select_one("h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()

    date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        raw = meta_date["content"].strip()
        m = re.match(r"(\d{4})-(\d{1,2})-(\d{1,2})", raw)
        if m:
            date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"

    body_text = ""
    content = soup.select_one(".j-fontContent, .ls-article-info")
    if content:
        for s in content.find_all("script"):
            s.decompose()
        # Use get_text() for all text within content div
        body_text = content.get_text(separator="\n", strip=True)

    return {"title": title, "date": date, "body_text": body_text}


def main():
    parser = argparse.ArgumentParser(description="淮北煤化工基地通知公告")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数(默认5)")
    parser.add_argument("--push", action="store_true", default=False)
    args = parser.parse_args()

    print(f"{SITE_NAME} 通知公告 — 爬取 {args.pages} 页", flush=True)

    all_items = []
    for page_no in range(args.pages):
        items = fetch_list(page_no)
        if not items:
            print(f"第{page_no+1}页: 无数据", flush=True)
            break
        print(f"第{page_no+1}页: {len(items)} 条", flush=True)
        all_items.extend(items)

    print(f"\n共获取 {len(all_items)} 条列表项", flush=True)

    details = []
    for idx, item in enumerate(all_items, 1):
        print(f"  [{idx}/{len(all_items)}] {item['title'][:35]}...", end=" ", flush=True)
        detail = fetch_detail(item["url"])
        if not detail:
            print("❌", flush=True)
            continue
        details.append({
            "site_name": SITE_NAME,
            "source_url": item["url"],
            "url": item["url"],
            "title": detail["title"] or item["title"],
            "pub_date": detail["date"],
            "summary": (detail["body_text"] or "")[:500],
            "content": detail["body_text"] or "",
            "category": GROUP,
            "tags": INDUSTRY,
            "attachments": "",
        })
        print(f"✅ {len(detail['body_text'])}字", flush=True)

    saved, skipped = len(details), len(all_items) - len(details)
    print(f"\n完成: {saved} 条, {skipped} 条失败", flush=True)

    if args.push and details:
        push_to_searchdb(details, batch_label=f"huaibei_tzgg_{args.pages}p")

    output = {"saved": saved, "skipped": skipped, "total": len(all_items), "push": args.push}
    print(f"\n---STATS---\n{json.dumps(output)}")


if __name__ == "__main__":
    main()
