#!/usr/bin/env python3
"""
crawl_ptxy_hjsp.py - 莆田市秀屿区人民政府-项目环境影响审批
===========================================================
站点: https://www.ptxy.gov.cn/zwgk/hjbh/xmhjyxsp/
CMS:  莆田21pt (TRS+avalon.js), API驱动
列表: /fjdzapp/search?channelid=201988&classsql=chnlid=26312&page=N&perpage=15
详情: TRS_Editor div

用法:
  python3 crawl_ptxy_hjsp.py              # 默认5页
  python3 crawl_ptxy_hjsp.py --pages 3    # 爬3页
  python3 crawl_ptxy_hjsp.py --dry-run    # 仅预览
"""

import sys, os, re, json, requests, time
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "https://www.ptxy.gov.cn"
LIST_API = f"{BASE_URL}/fjdzapp/search"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Referer": f"{BASE_URL}/zwgk/hjbh/xmhjyxsp/",
}

DEFAULT_PAGES = 5
PER_PAGE = 15


def log(msg):
    ts = datetime.now().strftime("%H:%M:%S")
    print(f"[{ts}] {msg}", flush=True)


def clean_title(title):
    """清洗标题：strip &middot; &nbsp; 等HTML实体前缀"""
    if not title:
        return title
    # 解码HTML实体
    title = title.replace("&middot;", "·").replace("&nbsp;", " ").replace("&amp;", "&")
    # Strip前缀空白/特殊字符
    title = title.strip("·. \t\n\r")
    return title.strip()


def clean_html(content_html):
    """清洗正文HTML：去冗余标签，保留表格HTML"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")

    # 移除style/script标签
    for tag in soup.find_all(["style", "script"]):
        tag.decompose()

    # 用<br>换行
    for br in soup.find_all("br"):
        br.replace_with("\n")

    # 递归提取内容：表格保留HTML，其他取文本
    def extract_content(element, is_inline=False):
        """is_inline=True时用空格拼接内联子元素，False时用换行拼接块级子元素"""
        parts = []
        for child in element.children:
            if child.name is None:
                # 文本节点
                text = str(child).strip()
                if text:
                    parts.append(text)
            elif child.name == "table":
                # 表格保留完整HTML
                parts.append(str(child))
            elif child.name in ("p", "div", "td", "th", "li", "section", "tr", "tbody",
                                "thead", "h1", "h2", "h3", "h4", "h5", "h6"):
                # 块级元素递归处理（块级→块级用\n\n分隔）
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            elif child.name in ("span", "font", "b", "i", "strong", "em", "a",
                                "u", "sub", "sup", "small", "label"):
                # 内联元素递归处理（内联级用空格拼接）
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            else:
                # 其他标签取文本
                text = child.get_text(strip=True)
                if text:
                    parts.append(text)
        if is_inline:
            # 内联上下文中用空格拼接（防止"联 系 人"中每个字断行）
            result = " ".join(p for p in parts if p.strip())
            # 合并多余空格
            result = re.sub(r" {2,}", " ", result)
        else:
            # 块级上下文中用\n\n分段
            result = "\n\n".join(p for p in parts if p.strip())
        return result.strip()

    result = extract_content(soup)
    return result.strip()


def fetch_list(page=1):
    """获取列表页API数据"""
    params = {
        "channelid": "201988",
        "classsql": "chnlid=26312",
        "sortfield": "-docorderpri,-docreltime",
        "page": page,
        "perpage": str(PER_PAGE),
    }
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        data = resp.json()
        if data.get("error"):
            log(f"API返回错误: {data}")
            return []
        return data.get("data", [])
    except Exception as e:
        log(f"列表页请求失败 (page={page}): {e}")
        return []


def fetch_detail(url):
    """获取详情页HTML并提取正文"""
    if not url.startswith("http"):
        url = BASE_URL + url
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"详情页请求失败 {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # 标题 - 优先从meta取
    title = None
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).split("_")[0].strip()

    title = clean_title(title) if title else ""

    # 发布日期
    pub_date = None
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # 正文 - TRS_Editor
    content_div = soup.find("div", class_="TRS_Editor")
    content_html = ""
    if content_div:
        content_html = str(content_div)

    return title, pub_date, content_html


def crawl(pages=DEFAULT_PAGES, dry_run=False):
    """主爬取逻辑"""
    all_items = []
    total_new = 0

    for page in range(1, pages + 1):
        log(f"爬取第 {page} 页...")
        items = fetch_list(page)
        if not items:
            log(f"第 {page} 页无数据，停止")
            break

        log(f"  获取到 {len(items)} 条")

        for item in items:
            title = clean_title(item.get("doctitle", ""))
            url = item.get("chnldocurl", item.get("docpuburl", "")).strip()
            # 相对URL转绝对URL
            if url.startswith("/"):
                url = BASE_URL + url
            date = (item.get("docreltime", "") or "")[:10]

            if dry_run:
                all_items.append({
                    "title": title,
                    "date": date,
                    "url": url,
                })
                continue

            # 获取详情
            log(f"  详情: {title[:40]}...")
            detail_title, detail_date, content_html = fetch_detail(url)

            # 合并标题
            final_title = detail_title or title
            final_date = detail_date or date

            # 清洗正文
            cleaned_content = clean_html(content_html) if content_html else ""

            all_items.append({
                "title": final_title,
                "date": final_date,
                "url": url,
                "content": cleaned_content,
            })
            total_new += 1

            time.sleep(0.5)  # 礼貌延迟

    return all_items, total_new


def push_to_searchdb(items, source_name="ptxy_hjsp", group_name="福建省",
                     industry="政府公告"):
    """入库到 search.db"""
    if not items:
        log("无数据可入库")
        return

    db_path = "/mnt/data/search.db"
    import subprocess

    inserted = 0
    for item in items:
        title = item.get("title", "")
        date = item.get("date", "")
        url = item.get("url", "")
        content = item.get("content", "")

        if not title or not url:
            continue

        # 使用 subprocess + sqlite3 CLI 写库（防WAL损坏）
        esc_title = title.replace("'", "''")
        esc_url = url.replace("'", "''")
        esc_content = content.replace("'", "''")
        esc_source = source_name.replace("'", "''")
        esc_group = group_name.replace("'", "''")
        esc_industry = industry.replace("'", "''")
        esc_date = date.replace("'", "''")

        sql = (
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, source_url, page_url, title, publish_date, summary, content, group_name, industry) "
            "VALUES ("
            f"'{esc_source}', "
            f"'{esc_url}', "
            f"'{esc_url}', "
            f"'{esc_title}', "
            f"'{esc_date}', "
            f"'{esc_title}', "
            f"'{esc_content}', "
            f"'{esc_group}', "
            f"'{esc_industry}'"
            ");"
        )
        cmd = ["sqlite3", db_path, f"PRAGMA busy_timeout=15000; {sql}"]
        try:
            result = subprocess.run(
                cmd, capture_output=True, text=True, timeout=30
            )
            if result.returncode == 0:
                inserted += 1
        except Exception as e:
            log(f"入库失败 {title[:20]}: {e}")

    log(f"入库完成: {inserted}/{len(items)} 条")

    # FTS同步
    log("同步FTS...")
    fts_sql = (
        "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r "
        f"WHERE r.site_name='{esc_source}' "
        "AND NOT EXISTS (SELECT 1 FROM gov_search s WHERE s.rowid=r.id);"
    )
    try:
        subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", db_path, fts_sql],
            capture_output=True, text=True, timeout=30
        )
    except Exception as e:
        log(f"FTS同步失败: {e}")


def main():
    import argparse
    parser = argparse.ArgumentParser(description="莆田秀屿-项目环境影响审批爬虫")
    parser.add_argument("--pages", type=int, default=DEFAULT_PAGES,
                        help=f"爬取页数 (默认: {DEFAULT_PAGES})")
    parser.add_argument("--dry-run", action="store_true",
                        help="仅预览不实际爬取详情")
    args = parser.parse_args()

    log(f"🚀 爬虫启动: ptxy_hjsp, pages={args.pages}, dry_run={args.dry_run}")

    items, total = crawl(pages=args.pages, dry_run=args.dry_run)

    if args.dry_run:
        log(f"\n=== 预览: 共 {len(items)} 条 ===")
        for item in items[:5]:
            print(f"  [{item['date']}] {item['title'][:50]}")
            print(f"  {item['url']}")
            print()
        return

    log(f"\n=== 爬取完成: {total} 条 ===")
    for item in items[:3]:
        print(f"  [{item['date']}] {item['title'][:50]}")
        print(f"  URL: {item['url']}")
        print(f"  正文: {len(item.get('content',''))} chars")
        print()

    # 正式入库
    push_to_searchdb(items)
    log("✅ 全部完成")


if __name__ == "__main__":
    main()
