#!/usr/bin/env python3
"""
菏泽市生态环境局 - 通知公告爬虫
URL: http://hzsthj.heze.gov.cn/stjlby/?catas=1590962981401923584
CMS: 菏泽ELS系统 (Vue + Axios + Spring Boot)
API: POST /els-service/article/{page}/{size}
详情: /{dwid}/{xxid}.html (正文内嵌在JS var memo 中)
共约 44 页 x 10 条 = 440 条
日跑: --max-pages 1 (增量取最新页)
"""

import sys, os, re, time, sqlite3, argparse, json
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

import warnings
warnings.filterwarnings("ignore")

BASE_URL = "http://hzsthj.heze.gov.cn"
SITE_NAME = "菏泽市生态环境局"
CATEGORY = "通知公告"
GROUP = "县区"
DB_PATH = "/mnt/data/search.db"
DWID = "2c908088819842f701819a2925400026"
CATAS = "1590962981401923584"
PAGE_SIZE = 10

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Content-Type": "application/json;charset=UTF-8",
}


def fetch_list(page):
    """通过 API 获取文章列表"""
    url = f"{BASE_URL}/els-service/article/{page}/{PAGE_SIZE}"
    data = {
        "type": [1],
        "fwzt": "3",
        "order": "fwdate",
        "catas": [CATAS],
        "dw": [DWID],
    }
    items = []
    try:
        resp = requests.post(url, headers=HEADERS, json=data, timeout=30)
        resp.raise_for_status()
        result = resp.json()
        if result.get("success") and result.get("data"):
            d = result["data"]
            for c in d.get("contents", []):
                xxid = c.get("xxid", "")
                subject = (c.get("subject") or "").strip()
                fwdate = (c.get("fwdate") or "").strip()
                if not xxid or not subject:
                    continue
                detail_url = f"{BASE_URL}/{DWID}/{xxid}.html"
                items.append({
                    "url": detail_url,
                    "title": subject,
                    "date": fwdate,
                    "xxid": xxid,
                })
        return items, d.get("totalPages", 0) if result.get("data") else 0
    except Exception as e:
        print(f"  [错误] 列表页 {page} 请求失败: {e}", flush=True)
        return [], 0


def fetch_detail(detail_url, xxid=None):
    """从详情页 JS 提取 memo 正文"""
    try:
        resp = requests.get(detail_url, headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        }, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text

        # 从 JS 中提取 memo 变量 (包含完整 HTML 正文)
        idx = html.find('var memo =')
        if idx < 0:
            return ""

        start = html.index('"', idx + 10) + 1
        end = start
        while end < len(html):
            if html[end] == '"':
                if end > start and html[end-1] == '\\':
                    end += 1
                    continue
                # 接受 "; 或 " 后跟空白/换行/文件结尾
                if end + 1 >= len(html) or html[end+1] in (';', ' ', '\n', '\r', '\t'):
                    memo_raw = html[start:end]
                    break
            end += 1
        else:
            return ""

        # 转义 unicode
        memo = re.sub(r'\\u([0-9a-fA-F]{4})', lambda m: chr(int(m.group(1), 16)), memo_raw)
        # 转义其他
        memo = memo.replace('\\"', '"')
        memo = memo.replace('\\/', '/')

        # 清理多余样式属性
        soup = BeautifulSoup(memo, "html.parser")

        # 处理图片和附件链接
        for img in soup.find_all("img"):
            src = img.get("src", "")
            if src and not src.startswith("http"):
                # 尝试补全路径
                if src.startswith("/"):
                    img["src"] = f"{BASE_URL}{src}"
                else:
                    img["src"] = f"{BASE_URL}/{src}"
        for a in soup.find_all("a"):
            href = a.get("href", "")
            if href and not href.startswith("http") and not href.startswith("javascript"):
                if href.startswith("/"):
                    a["href"] = f"{BASE_URL}{href}"
                else:
                    a["href"] = f"{BASE_URL}/{href}"

        return clean_content_html(str(soup))

    except Exception as e:
        print(f"  [错误] 详情页获取失败 {detail_url}: {e}", flush=True)
        return ""


def clean_content_html(html):
    if not html:
        return ""
    cleaned = re.sub(r'\sstyle="[^"]*"', '', html)
    cleaned = re.sub(r'<span[^>]*>|</span>', '', cleaned)
    cleaned = re.sub(r'<strong[^>]*>|</strong>', '', cleaned)
    cleaned = re.sub(r'<b[^>]*>|</b>', '', cleaned)
    cleaned = re.sub(r'<font[^>]*>|</font>', '', cleaned)
    cleaned = re.sub(r'<o:p[^>]*>|</o:p>', '', cleaned)
    cleaned = re.sub(r'\n{3,}', '\n\n', cleaned)
    return cleaned.strip()


def crawl(max_pages=None):
    s = requests.Session()
    s.headers.update(HEADERS)

    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    existing = set()
    cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    for row in cur:
        existing.add(row[0])

    print(f"已有 {len(existing)} 条菏泽记录", flush=True)

    # 先取第一页确定总页数
    first_items, total_pages = fetch_list(1)
    if not first_items:
        print("无法获取列表数据", flush=True)
        db.close()
        return 0

    total_pages = int(total_pages) if total_pages else 44
    pages_to_fetch = min(max_pages, total_pages) if max_pages else 5
    print(f"共约 {total_pages} 页, 爬取前 {pages_to_fetch} 页", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"[第 {page}/{pages_to_fetch} 页] ", end="", flush=True)
        items = first_items if page == 1 else fetch_list(page)[0]
        if not items:
            print("无数据", flush=True)
            continue

        page_new = 0
        for item in items:
            full_url = item["url"]
            if full_url in existing:
                total_skip += 1
                continue

            title = item["title"]
            publish_date = item["date"] if len(item["date"]) >= 10 else ""

            if not title:
                total_skip += 1
                continue

            # 正文
            body = fetch_detail(full_url, item.get("xxid"))
            if not body or len(body.strip()) < 50:
                body = f"<p>详情见原文：<a href='{full_url}'>{title}</a></p>"

            summary = re.sub(r'<[^>]+>', ' ', body)
            summary = re.sub(r'\s+', ' ', summary).strip()[:300]
            has_table = 1 if '<table' in body else 0

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, has_table)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, full_url, full_url, title[:500], publish_date,
                    summary[:500], body, 'active', CATEGORY, GROUP, has_table
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,)).fetchone()
                    if row:
                        rid = row[0]
                        try:
                            # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                            #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                            db.commit()
                            db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                                       (rid, title[:500], SITE_NAME, summary[:200]))
                        except sqlite3.IntegrityError:
                            pass
                    existing.add(full_url)
                    db.commit()
                    total_new += 1
                    page_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"x", end="", flush=True)
                total_error += 1

        print(f"+{page_new} 条 (累计 {total_new})", flush=True)
        if page < pages_to_fetch:
            time.sleep(0.5)

    db.close()
    print(f"\n完成! 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}", flush=True)
    return total_new


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
