#!/usr/bin/env python3
"""
烟台市生态环境局 — 环评公示栏目爬虫
URL: https://hbj.yantai.gov.cn/col/col23547/index.html
CMS: Hanweb 6.2.6.11 (API动态加载)
API: GET /api-gateway/jpaas-publish-server/front/page/build/unit
Detail: <div class="PageArticleContent"> → p content
"""

import requests
from bs4 import BeautifulSoup
import json
import re
import sys
import os
import argparse
from datetime import datetime

# ─── 配置 ─────────────────────────────────────────────
BASE_URL = "https://hbj.yantai.gov.cn"
LIST_API = "https://hbj.yantai.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_NAME = "烟台市生态环境局"
COLUMN_NAME = "环评公示"
GROUP = "山东省"
INDUSTRY = "环评公示"
ROWS = 15  # 每页条数

BASE_API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "168",
    "tplSetId": "wvUixQFOqDClxrcBNh9df",
    "pageType": "column",
    "tagId": "信息列表",
    "editType": "null",
    "pageId": "23547",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import fetch_page, push_to_searchdb


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def fetch_list_page(page_no: int) -> list:
    """获取列表页，返回 [(title, url, pub_date), ...]"""
    params = {**BASE_API_PARAMS}
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": ROWS}, ensure_ascii=False)
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=15)
        resp.raise_for_status()
        data = resp.json()
    except Exception as e:
        print(f"  ❌ 列表页 {page_no} 请求失败: {e}", flush=True)
        return []

    html = data.get("data", {}).get("html", "")
    if not html:
        print(f"  ❌ 列表页 {page_no} 无HTML数据", flush=True)
        return []

    soup = BeautifulSoup(html, "html.parser")
    items = []
    for tr in soup.find_all("tr"):
        tds = tr.find_all("td")
        if len(tds) >= 2:
            a_tag = tds[0].find("a")
            if a_tag and a_tag.get("href") and a_tag.get("title"):
                href = a_tag["href"].strip()
                title = a_tag["title"].strip()
                # 处理相对URL
                if href.startswith("/"):
                    url = BASE_URL + href
                else:
                    url = href
                date_str = tds[1].get_text(strip=True) if len(tds) > 1 else ""
                items.append((title, url, date_str))

    print(f"  📄 第{page_no}页: {len(items)}条", flush=True)
    return items


def fetch_detail(url: str) -> tuple:
    """
    获取详情页，返回 (title, content, publish_date)
    返回 (None, ..., ...) 表示失败
    """
    resp = fetch_page(url)
    if resp is None:
        return None, "", ""

    soup = BeautifulSoup(resp, "html.parser")

    # 标题: 优先 ArticleTitle meta
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True)

    # 日期
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()

    # 正文: <div class="PageArticleContent">
    content_div = soup.find("div", class_="PageArticleContent")
    if not content_div:
        # fallback
        content_div = soup.find("div", class_="wzy_bd") or \
                      soup.find("div", id="zoomFont") or \
                      soup.find("div", class_="content")

    content = ""
    if content_div:
        # 移除script/style
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()
        # 递归提取正文但保留表格HTML，修复内联span多余换行
        content_parts = []
        skip_classes = {"ArticleTitle", "ArticleAttr"}  # 元数据div跳过
        def walk_element(el):
            for child in el.children:
                if child.name == "table":
                    content_parts.append(str(child))
                elif child.name and child.name not in ("script", "style"):
                    # 跳过元数据div
                    child_classes = set(child.get("class", []) or [])
                    if child_classes & skip_classes:
                        continue
                    if child.find_all("table"):
                        walk_element(child)
                    else:
                        # p标签内联span合并（不用\n分隔），其他块级元素用空格
                        sep = "" if child.name == "p" else " "
                        txt = child.get_text(sep, strip=True)
                        if txt:
                            content_parts.append(txt)
        walk_element(content_div)
        content = "\n".join(content_parts) if content_parts else body_text(content_div)

    return title, content, pub_date


def get_total_count() -> int:
    """获取总条数"""
    params = {**BASE_API_PARAMS}
    params["paramJson"] = json.dumps({"pageNo": 1, "pageSize": ROWS}, ensure_ascii=False)
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=15)
        resp.raise_for_status()
        data = resp.json()
        html = data.get("data", {}).get("html", "")
        m = re.search(r'count="(\d+)"', html)
        if m:
            return int(m.group(1))
    except Exception:
        pass
    return 0


def crawl(pages: int = 5, push: bool = True):
    """主爬虫"""
    total = get_total_count()
    total_pages = max(1, (total + ROWS - 1) // ROWS) if total else 7
    actual_pages = min(pages, total_pages) if pages > 0 else total_pages

    print(f"🔍 {SITE_NAME} - {COLUMN_NAME}", flush=True)
    print(f"📊 总计 {total} 条, {total_pages} 页, 本次爬取 {actual_pages} 页", flush=True)

    all_items = []
    for page in range(1, actual_pages + 1):
        items = fetch_list_page(page)
        if not items:
            print(f"  ⚠️ 第{page}页为空，停止翻页", flush=True)
            break
        all_items.extend(items)

    print(f"\n📋 列表共 {len(all_items)} 条", flush=True)
    print(f"{'='*60}", flush=True)

    push_items = []
    fail_count = 0

    for idx, (title, url, date_str) in enumerate(all_items, 1):
        print(f"\n[{idx}/{len(all_items)}] {title[:60]}", flush=True)
        print(f"    📎 {url}", flush=True)
        print(f"    📅 {date_str}", flush=True)

        # 取详情
        detail_title, content, pub_date = fetch_detail(url)
        if detail_title is None:
            print(f"    ❌ 详情获取失败", flush=True)
            fail_count += 1
            continue

        final_title = detail_title or title
        final_date = pub_date or date_str

        content_preview = content[:80].replace('\n', ' ')
        print(f"    ✅ 正文 {len(content)} 字: {content_preview}...", flush=True)

        push_items.append({
            "title": final_title,
            "url": url,
            "source_url": url,
            "pub_date": final_date,
            "content": content,
            "summary": content[:500] if content else final_title,
            "site_name": SITE_NAME,
            "group_name": GROUP,
            "industry": INDUSTRY,
            "category": COLUMN_NAME,
        })

    # 批量推送到search.db
    if push and push_items:
        print(f"\n{'='*60}", flush=True)
        print(f"📤 正在推送 {len(push_items)} 条到 search.db...", flush=True)
        push_to_searchdb(push_items, batch_label=f"{SITE_NAME}-{COLUMN_NAME}")
    elif push:
        print(f"\n⚠️ 无数据可推送", flush=True)

    print(f"\n{'='*60}", flush=True)
    print(f"✅ 完成: 共{len(push_items)}条, 失败{fail_count}", flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=f"{SITE_NAME} - {COLUMN_NAME} 爬虫")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数 (0=全部)")
    parser.add_argument("--no-push", action="store_true", help="不推送到search.db")
    args = parser.parse_args()

    crawl(pages=args.pages, push=not args.no_push)
