#!/usr/bin/env python3
"""
舟山市生态环境局 — 建设项目环境影响环评审批公示栏目爬虫
URL: http://zshbj.zhoushan.gov.cn/col/col1229856873/index.html
CMS: Hanweb 4.5.12.1 (Zhejiang variant, API动态加载)
需自定义UA头绕过OpenResty WAF (403)
API: GET /api-gateway/jpaas-publish-server/front/page/build/unit
Detail: <div id="div_content"> / <div id="div_border"> 含表格
"""

import requests
from bs4 import BeautifulSoup
import json
import re
import sys
import os
import argparse
from datetime import datetime

# ─── 配置 ─────────────────────────────────────────────
BASE_URL = "http://zshbj.zhoushan.gov.cn"
LIST_API = "http://zshbj.zhoushan.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
COL_LIST_URL = "http://zshbj.zhoushan.gov.cn/col/col1229856873/index.html"
SITE_NAME = "舟山市生态环境局"
COLUMN_NAME = "建设项目环境影响环评审批公示"
GROUP = "浙江省"
INDUSTRY = "环评公示"
ROWS = 16  # 每页条数

BASE_API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "3122",
    "tplSetId": "LS880IEmCMko8ZoNROB1y",
    "pageType": "column",
    "tagId": "栏目内容list",
    "editType": "null",
    "pageId": "1229856873",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
# 自定义fetch_page，因为原版的不带Accept-Language头会403
def safe_fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        if r.status_code == 200:
            r.encoding = 'utf-8'
            return r.text
        print(f"    ⚠️ HTTP {r.status_code}", flush=True)
        return None
    except Exception as e:
        print(f"    ❌ 请求失败: {e}", flush=True)
        return None


def fetch_list_page(page_no: int) -> list:
    """获取列表页，返回 [(title, url, pub_date), ...]"""
    params = {**BASE_API_PARAMS}
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": ROWS}, ensure_ascii=False)
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=15)
        resp.raise_for_status()
        data = resp.json()
    except Exception as e:
        print(f"  ❌ 列表页 {page_no} 请求失败: {e}", flush=True)
        return []

    html = data.get("data", {}).get("html", "")
    if not html:
        print(f"  ❌ 列表页 {page_no} 无HTML数据", flush=True)
        return []

    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a_tag in soup.find_all("a", href=True):
        href = a_tag["href"].strip()
        title = a_tag.get("title", "").strip() or a_tag.get_text(strip=True)
        if not title or not href.startswith("/"):
            continue
        # 找相邻的日期
        date_str = ""
        parent = a_tag.parent
        if parent:
            # 查找父级tr中的其他td/span
            date_el = parent.find_previous("span", style=lambda x: x and "color" in str(x).lower())
            if not date_el:
                date_el = parent.parent.find_all("td")
                if len(date_el) >= 2:
                    # Last td usually contains date
                    date_str = date_el[-1].get_text(strip=True)
            if date_el and date_el != a_tag.parent:
                date_str = date_el.get_text(strip=True)
            if not date_str:
                # Try to find date in siblings or within same row
                row = parent
                while row and row.name != "tr":
                    row = row.parent
                if row:
                    tds = row.find_all("td")
                    if len(tds) >= 2:
                        date_str = tds[-1].get_text(strip=True)

        url = BASE_URL + href
        items.append((title, url, date_str))

    print(f"  📄 第{page_no}页: {len(items)}条", flush=True)
    return items


def extract_content_with_tables(content_div):
    """递归提取正文，保留表格HTML，跳过元数据div"""
    content_parts = []
    skip_classes = set()  # This site doesn't have standard metadata divs

    def walk_element(el):
        for child in el.children:
            if child.name == "table":
                content_parts.append(str(child))
            elif child.name and child.name not in ("script", "style"):
                child_classes = set(child.get("class", []) or [])
                if child_classes & skip_classes:
                    continue
                if child.find_all("table"):
                    walk_element(child)
                else:
                    sep = "" if child.name == "p" else " "
                    txt = child.get_text(sep, strip=True)
                    if txt:
                        content_parts.append(txt)
    walk_element(content_div)
    return "\n".join(content_parts) if content_parts else content_div.get_text(" ", strip=True)


def fetch_detail(url: str) -> tuple:
    """获取详情页，返回 (title, content, publish_date)"""
    html = safe_fetch(url)
    if html is None:
        return None, "", ""

    soup = BeautifulSoup(html, "html.parser")

    # 标题: ArticleTitle meta
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        titletag = soup.find("title")
        if titletag:
            title = titletag.get_text(strip=True)

    # 日期
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()

    # 正文: 优先 div#div_content (纯内容含表格), 次选 div#div_border
    content_div = soup.find("div", id="div_content") or \
                  soup.find("div", id="div_border") or \
                  soup.find("div", id="div_main_content")
    content = extract_content_with_tables(content_div) if content_div else ""

    return title, content, pub_date


def get_total_count() -> int:
    """获取总条数"""
    params = {**BASE_API_PARAMS}
    params["paramJson"] = json.dumps({"pageNo": 1, "pageSize": ROWS}, ensure_ascii=False)
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=15)
        data = resp.json()
        html = data.get("data", {}).get("html", "")
        m = re.search(r'count="(\d+)"', html)
        if m:
            return int(m.group(1))
    except Exception:
        pass
    return 0


def crawl(pages: int = 5, push: bool = True):
    """主爬虫"""
    total = get_total_count()
    total_pages = max(1, (total + ROWS - 1) // ROWS) if total else 37
    actual_pages = min(pages, total_pages) if pages > 0 else total_pages

    print(f"🔍 {SITE_NAME} - {COLUMN_NAME}", flush=True)
    print(f"📊 总计 {total} 条, {total_pages} 页, 本次爬取 {actual_pages} 页", flush=True)

    all_items = []
    for page in range(1, actual_pages + 1):
        items = fetch_list_page(page)
        if not items:
            print(f"  ⚠️ 第{page}页为空，停止翻页", flush=True)
            break
        all_items.extend(items)

    print(f"\n📋 列表共 {len(all_items)} 条", flush=True)
    print(f"{'='*60}", flush=True)

    push_items = []
    fail_count = 0

    for idx, (title, url, date_str) in enumerate(all_items, 1):
        print(f"\n[{idx}/{len(all_items)}] {title[:60]}", flush=True)
        print(f"    📎 {url}", flush=True)
        if date_str:
            print(f"    📅 {date_str}", flush=True)

        detail_title, content, pub_date = fetch_detail(url)
        if detail_title is None:
            print(f"    ❌ 详情获取失败", flush=True)
            fail_count += 1
            continue

        final_title = detail_title or title
        final_date = pub_date or date_str

        content_preview = content[:80].replace('\n', ' ')
        print(f"    ✅ 正文 {len(content)} 字: {content_preview}...", flush=True)

        push_items.append({
            "title": final_title,
            "url": url,
            "source_url": url,
            "pub_date": final_date,
            "content": content,
            "summary": content[:500] if content else final_title,
            "site_name": SITE_NAME,
            "group_name": GROUP,
            "industry": INDUSTRY,
            "category": COLUMN_NAME,
        })

    if push and push_items:
        print(f"\n{'='*60}", flush=True)
        print(f"📤 正在推送 {len(push_items)} 条到 search.db...", flush=True)
        push_to_searchdb(push_items, batch_label=f"{SITE_NAME}-{COLUMN_NAME}")
    elif push:
        print(f"\n⚠️ 无数据可推送", flush=True)

    print(f"\n{'='*60}", flush=True)
    print(f"✅ 完成: 共{len(push_items)}条, 失败{fail_count}", flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=f"{SITE_NAME} - {COLUMN_NAME} 爬虫")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数 (0=全部)")
    parser.add_argument("--no-push", action="store_true", help="不推送到search.db")
    args = parser.parse_args()

    crawl(pages=args.pages, push=not args.no_push)
