#!/usr/bin/env python3
"""
crawl_bachu_sthj.py - 巴楚县人民政府-生态环境
=============================================
站点: https://www.bachu.gov.cn/bcx/c115734/zwgk.shtml
CMS:  巴楚政务CMS (jQuery)
列表: zwgk_N.shtml (共3页，每页15条)
详情: <div id="zoomcon" class="xxgk-tt-content xxgk-tt-content-body">

用法:
  python3 crawl_bachu_sthj.py              # 默认5页
  python3 crawl_bachu_sthj.py --pages 3    # 爬3页
  python3 crawl_bachu_sthj.py --dry-run    # 仅预览
"""

import sys, os, re, json, requests, time
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "https://www.bachu.gov.cn"
LIST_PATH = "/bcx/c115734/zwgk.shtml"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

DEFAULT_PAGES = 5


def log(msg):
    ts = datetime.now().strftime("%H:%M:%S")
    print(f"[{ts}] {msg}", flush=True)


def clean_title(title):
    """清洗标题：strip &middot; &nbsp; 等HTML实体前缀"""
    if not title:
        return ""
    title = title.replace("&middot;", "·").replace("&nbsp;", " ").replace("&amp;", "&")
    return title.strip("·. \t\n\r")


def clean_html(content_html):
    """清洗正文HTML：去冗余标签，保留表格HTML"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")

    for tag in soup.find_all(["style", "script"]):
        tag.decompose()
    for br in soup.find_all("br"):
        br.replace_with("\n")
    for tag in soup.find_all(["o:p"]):
        tag.decompose()

    # 修复空<p>/包裹型<div>嵌套(只含<p>子元素)
    for tag in list(soup.find_all(["p", "div"])):
        kids = [c for c in tag.children if not (c.name is None and not str(c).strip())]
        if kids and all(c.name == "p" for c in kids):
            tag.unwrap()

    def extract_content(element, is_inline=False):
        parts = []
        for child in element.children:
            if child.name is None:
                text = str(child).strip()
                if text:
                    parts.append(text)
            elif child.name == "table":
                parts.append(str(child))
            elif child.name in ("p", "div", "td", "th", "li", "section", "tr", "tbody",
                                "thead", "h1", "h2", "h3", "h4", "h5", "h6"):
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            elif child.name in ("span", "font", "b", "i", "strong", "em", "a",
                                "u", "sub", "sup", "small", "label"):
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            else:
                text = child.get_text(strip=True)
                if text:
                    parts.append(text)
        if is_inline:
            result = " ".join(p for p in parts if p.strip())
            result = re.sub(r" {2,}", " ", result)
        else:
            result = "\n\n".join(p for p in parts if p.strip())
        return result.strip()

    result = extract_content(soup)

    # 后处理：对单<p>内嵌的中文章节标题做分割
    # 如 "...意见和建议。 一、建设基本情况" → "...意见和建议。\n\n一、建设基本情况"
    # 分割模式：一二三...、 （一）（二）...、 1. 2. 3.(跟中文)等
    section_pats = [
        r'(?<!\n)\s+(?=[一二三四五六七八九十]+[、．])',  # 一、 二、 三、 ...
        r'(?<!\n)\s+(?=[（][一二三四五六七八九十][）])',   # （一）（二）（三）...
        r'(?<=[。！？])\s+(?=\d+[.、]\s*[\u4e00-\u9fff])',  # 1. 2. 后跟中文
        r'(?<!\n)\s{2,}(?=[一二三四五六七八九十]+[、．])',  # 空格较多的版本
    ]
    for pat in section_pats:
        result = re.sub(pat, '\n\n', result)

    return result.strip()


def fetch_list(page=1):
    """获取列表页"""
    if page == 1:
        url = f"{BASE_URL}{LIST_PATH}"
    else:
        url = f"{BASE_URL}/bcx/c115734/zwgk_{page}.shtml"

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"列表页请求失败 (page={page}): {e}")
        return [], 0

    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="innerList")
    items = []
    if ul:
        for li in ul.find_all("li", recursive=False):
            a_tag = li.find("a")
            span_tag = li.find("span", class_="fr")
            if a_tag and span_tag:
                href = a_tag.get("href", "")
                if href and not href.startswith("http"):
                    href = BASE_URL + href
                title = clean_title(a_tag.get("title", a_tag.get_text(strip=True)))
                date = span_tag.get_text(strip=True)
                items.append({"title": title, "date": date, "url": href})

    return items


def fetch_detail(url):
    """获取详情页"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"详情页请求失败 {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # 标题
    title = None
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).split("_")[0].strip()
    title = clean_title(title) if title else ""

    # 日期
    pub_date = None
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # 正文 - zoomcon（用 encode_contents() 取内部HTML，避免外层标签导致分段丢失）
    content_html = ""
    zoom_div = soup.find("div", id="zoomcon")
    if zoom_div:
        content_html = zoom_div.encode_contents().decode("utf-8")
    else:
        # fallback: try class xxgk-tt-content
        fallback = soup.find("div", class_="xxgk-tt-content")
        if fallback:
            content_html = fallback.encode_contents().decode("utf-8")

    return title, pub_date, content_html


def push_to_searchdb(items, source_name="bachu_sthj", group_name="新疆",
                     industry="政府公告"):
    """入库到 search.db"""
    if not items:
        log("无数据可入库")
        return

    db_path = "/mnt/data/search.db"
    import subprocess

    inserted = 0
    for item in items:
        title = item.get("title", "")
        date = item.get("date", "")
        url = item.get("url", "")
        content = item.get("content", "")

        if not title or not url:
            continue

        esc_title = title.replace("'", "''")
        esc_url = url.replace("'", "''")
        esc_source = source_name.replace("'", "''")
        esc_group = group_name.replace("'", "''")
        esc_industry = industry.replace("'", "''")
        esc_date = date.replace("'", "''")

        esc_content = content.replace("'", "''") if content else ""

        use_sql = (
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, source_url, page_url, title, publish_date, summary, content, group_name, industry) "
            "VALUES ("
            f"'{esc_source}', "
            f"'{esc_url}', "
            f"'{esc_url}', "
            f"'{esc_title}', "
            f"'{esc_date}', "
            f"'{esc_title}', "
            f"'{esc_content}', "
            f"'{esc_group}', "
            f"'{esc_industry}'"
            ");"
        )

        cmd = ["sqlite3", db_path, f"PRAGMA busy_timeout=15000; {use_sql}"]
        try:
            result = subprocess.run(
                cmd, capture_output=True, text=True, timeout=30
            )
            if result.returncode == 0:
                inserted += 1
        except Exception:
            pass

    log(f"入库完成: {inserted}/{len(items)} 条")

    # FTS同步
    log("同步FTS...")
    fts_sql = (
        "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r "
        f"WHERE r.site_name='{esc_source}' "
        "AND NOT EXISTS (SELECT 1 FROM gov_search s WHERE s.rowid=r.id);"
    )
    try:
        subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", db_path, fts_sql],
            capture_output=True, text=True, timeout=30
        )
    except Exception as e:
        log(f"FTS同步失败: {e}")


def main():
    import argparse
    parser = argparse.ArgumentParser(description="巴楚县-生态环境爬虫")
    parser.add_argument("--pages", type=int, default=DEFAULT_PAGES,
                        help=f"爬取页数 (默认: {DEFAULT_PAGES})")
    parser.add_argument("--dry-run", action="store_true",
                        help="仅预览不实际爬取详情")
    args = parser.parse_args()

    log(f"🚀 爬虫启动: bachu_sthj, pages={args.pages}, dry_run={args.dry_run}")

    all_items = []
    for page in range(1, args.pages + 1):
        log(f"爬取第 {page} 页...")
        items = fetch_list(page)
        if not items:
            log(f"第 {page} 页无数据，停止")
            break
        log(f"  获取到 {len(items)} 条")

        for item in items:
            if args.dry_run:
                all_items.append(item)
                continue

            log(f"  详情: {item['title'][:40]}...")
            detail_title, detail_date, content_html = fetch_detail(item["url"])
            final_title = detail_title or item["title"]
            final_date = detail_date or item["date"]
            cleaned_content = clean_html(content_html) if content_html else ""

            all_items.append({
                "title": final_title,
                "date": final_date,
                "url": item["url"],
                "content": cleaned_content,
            })
            time.sleep(0.5)

    log(f"\n=== 爬取完成: {len(all_items)} 条 ===")
    for item in all_items[:3]:
        print(f"  [{item['date']}] {item['title'][:50]}")
        print(f"  URL: {item['url']}")
        if not args.dry_run:
            print(f"  正文: {len(item.get('content', ''))} chars")
        print()

    if not args.dry_run:
        push_to_searchdb(all_items)

    log("✅ 全部完成")


if __name__ == "__main__":
    main()
