#!/usr/bin/env python3
"""
牡丹江市人民政府 - 环境影响评价栏目爬虫
URL: https://mdj.gov.cn/mdjsrmzf/c100041/zfxxgk_list.shtml
CMS: 牡丹江站群系统 (JSON API + ucapcontent/TRS_Editor 详情页)
API: /search/{channelId}?page=N&_pageSize=15

配置:
  channelId = "1f7fbb73b2a141d080bad0cd5478586d"
  15条/页, 共1325条/89页

日跑: --max-pages 1 (增量)
"""

import sys, os, re, time, sqlite3, argparse
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

import warnings
warnings.filterwarnings("ignore")

BASE_URL = "https://mdj.gov.cn"
API_URL = f"{BASE_URL}/search/1f7fbb73b2a141d080bad0cd5478586d"
PAGE_SIZE = 15
SITE_NAME = "牡丹江市人民政府"
CATEGORY = "环境影响评价"
GROUP = "黑龙江"
DB_PATH = "/mnt/data/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE_URL}/mdjsrmzf/c100041/zfxxgk_list.shtml",
}


def clean_content_html(html):
    """清理正文HTML"""
    if not html:
        return ""
    cleaned = re.sub(r'\sstyle="[^"]*"', '', html)
    cleaned = re.sub(r'<span[^>]*>|</span>', '', cleaned)
    cleaned = re.sub(r'<strong[^>]*>|</strong>', '', cleaned)
    cleaned = re.sub(r'<b[^>]*>|</b>', '', cleaned)
    cleaned = re.sub(r'<font[^>]*>|</font>', '', cleaned)
    cleaned = re.sub(r'<o:p[^>]*>|</o:p>', '', cleaned)
    cleaned = re.sub(r'\n{3,}', '\n\n', cleaned)
    return cleaned.strip()


def fetch_list(page):
    """通过 API 获取列表页数据"""
    url = (f"{API_URL}?page={page}&_pageSize={PAGE_SIZE}"
           f"&_isAgg=true&_isJson=true&_template=index&_rangeTimeGte=&_channelName=")
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        data = resp.json()
        if "data" in data and "results" in data["data"]:
            return data["data"]["results"], int(data["data"]["total"])
        return [], 0
    except Exception as e:
        print(f"  [错误] 列表页 {page} 请求失败: {e}", flush=True)
        return [], 0


def fetch_detail(detail_url):
    """获取详情页正文"""
    full_url = urljoin(BASE_URL, detail_url)
    try:
        resp = requests.get(full_url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")

        # #zoomcon > ucapcontent (新样式)
        zoomcon = soup.select_one("#zoomcon")
        if zoomcon:
            ucap = zoomcon.select_one("ucapcontent")
            if ucap:
                content_html = str(ucap)
                tables = ucap.find_all("table")
                text_len = len(ucap.get_text(strip=True))
                if tables or text_len >= 400:
                    return _proc(content_html)
            else:
                # zoomcon 直接内容
                content_html = str(zoomcon)
                return _proc(content_html)

        # div.TRS_Editor (旧样式)
        trs = soup.select_one("div.TRS_Editor")
        if trs:
            return _proc(str(trs))

        # .article_content_body (备用)
        body = soup.select_one(".article_content_body")
        if body:
            return _proc(str(body))

        # 附件区
        appendix = soup.select_one(".article_appendix")
        if appendix:
            links = []
            for a in appendix.find_all("a"):
                href = a.get("href", "")
                txt = a.get_text(strip=True)
                if href and not href.startswith("javascript"):
                    links.append(f'<a href="{urljoin(BASE_URL, href)}">{txt}</a>')
            if links:
                return "<p>附件：</p>\n" + "\n".join(links)

        return ""
    except Exception as e:
        print(f"  [错误] 详情页获取失败 {detail_url}: {e}", flush=True)
        return ""


def _proc(html):
    """处理图片和链接 URL"""
    soup = BeautifulSoup(html, "html.parser")
    for img in soup.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith("http"):
            img["src"] = urljoin(BASE_URL, src)
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith("http") and not href.startswith("javascript"):
            a["href"] = urljoin(BASE_URL, href)
    return str(soup).strip()


def build_attachments(item):
    """构建附件Markdown"""
    parts = []
    for res in item.get("resList", []):
        fp = res.get("filePathNew", "") or res.get("filePath", "")
        fn = res.get("fileName", "") or res.get("title", "附件")
        if fp:
            fu = urljoin(BASE_URL, fp)
            parts.append(f'<a href="{fu}">{fn}</a>')
    return "\n\n".join(parts) if parts else ""


def crawl(max_pages=None):
    """主爬取逻辑"""
    s = requests.Session()
    s.headers.update(HEADERS)

    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    existing = set()
    cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    for row in cur:
        existing.add(row[0])

    print(f"已有 {len(existing)} 条牡丹江记录", flush=True)

    first_items, total = fetch_list(1)
    if total == 0:
        print("无法获取列表数据", flush=True)
        db.close()
        return 0

    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    pages_to_fetch = min(max_pages, total_pages) if max_pages else total_pages

    print(f"共 {total} 条 / {total_pages} 页, 爬取 {pages_to_fetch} 页", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"[第 {page}/{pages_to_fetch} 页] ", end="", flush=True)
        items = first_items if page == 1 else fetch_list(page)[0]
        if not items:
            print("无数据", flush=True)
            continue

        page_new = 0
        for item in items:
            url = item.get("url", "")
            if not url:
                continue
            full_url = urljoin(BASE_URL, url)
            if full_url in existing:
                total_skip += 1
                continue

            title = (item.get("title") or "").replace("&nbsp;", " ").strip()
            if not title:
                title = (item.get("subTitle") or "").replace("&nbsp;", " ").strip()
            if not title:
                total_skip += 1
                continue

            publish_date = ""
            ds = item.get("publishedTimeStr", "")
            if ds and len(ds) >= 10:
                publish_date = ds[:10]

            # 正文
            body = fetch_detail(url)

            # API contentHtml/content 兜底
            if not body or len(body.strip()) < 20:
                body = item.get("contentHtml", "") or item.get("content", "") or ""
                body = body.strip()

            # 附件
            atts = build_attachments(item)
            if atts:
                text_len = len(BeautifulSoup(body, "html.parser").get_text(strip=True)) if body else 0
                if text_len < 100:
                    body = f"<p>详情内容见附件：</p>\n{atts}"
                else:
                    body += f"\n\n<hr>\n<p><strong>附件：</strong></p>\n{atts}"

            body = clean_content_html(body)
            summary = re.sub(r'<[^>]+>', ' ', body)
            summary = re.sub(r'\s+', ' ', summary).strip()[:300]
            att_md = ""  # attachments already embedded in content

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, full_url, full_url, title[:500], publish_date,
                    summary[:500], body, 'active', CATEGORY, GROUP, att_md
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,)).fetchone()
                    if row:
                        rid = row[0]
                        try:
                            # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                            #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                            db.commit()
                            db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                                       (rid, title[:500], SITE_NAME, summary[:200]))
                        except sqlite3.IntegrityError:
                            pass
                    existing.add(full_url)
                    db.commit()
                    total_new += 1
                    page_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"x", end="", flush=True)
                total_error += 1

        print(f"+{page_new} 条 (累计 {total_new})", flush=True)

        if page < pages_to_fetch:
            time.sleep(0.5)

    db.close()
    print(f"\n完成! 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}", flush=True)
    return total_new


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
