#!/usr/bin/env python3
"""
crawl_ynsthjt_npzx.py - 云南省生态环境厅-拟审查项目公示
========================================================
站点: https://sthjt.yn.gov.cn/hpgl/jsxmhjyxpj/npzxmgs/wap.html
CMS:  云南政务WAP站 (Bootstrap + jQuery + TRS)
列表: 单页240条，ul.news-ul > li > a[href] + .news-ul-tt + .news-ul-date
详情: <div class="TRS_Editor">, <h1 class="article-tt">标题, 日期从.article-date提取

用法:
  python3 crawl_ynsthjt_npzx.py              # 默认全部
  python3 crawl_ynsthjt_npzx.py --pages 5    # 限制最多爬N条(X页×10条,实际全部在1页)
  python3 crawl_ynsthjt_npzx.py --dry-run    # 仅预览
"""

import sys, os, re, json, requests, time
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "https://sthjt.yn.gov.cn"
LIST_PATH = "/hpgl/jsxmhjyxpj/npzxmgs/wap.html"
DETAIL_DIR = "/hpgl/jsxmhjyxpj/npzxmgs"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

DEFAULT_PAGES = 5


def log(msg):
    ts = datetime.now().strftime("%H:%M:%S")
    print(f"[{ts}] {msg}", flush=True)


def clean_title(title):
    """清洗标题：strip &middot; &nbsp; 等HTML实体前缀"""
    if not title:
        return ""
    title = title.replace("&middot;", "·").replace("&nbsp;", " ").replace("&amp;", "&")
    return title.strip("·. \\t\\n\\r")


def clean_html(content_html):
    """清洗正文HTML：去冗余标签，保留表格HTML"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")

    for tag in soup.find_all(["style", "script"]):
        tag.decompose()
    for br in soup.find_all("br"):
        br.replace_with("\n")
    for tag in soup.find_all(["o:p"]):
        tag.decompose()

    # 修复空<p>/包裹型<div>嵌套(只含<p>子元素)
    for tag in list(soup.find_all(["p", "div"])):
        kids = [c for c in tag.children if not (c.name is None and not str(c).strip())]
        if kids and all(c.name == "p" for c in kids):
            tag.unwrap()

    def extract_content(element, is_inline=False):
        parts = []
        for child in element.children:
            if child.name is None:
                text = str(child).strip()
                if text:
                    parts.append(text)
            elif child.name == "table":
                parts.append(str(child))
            elif child.name in ("p", "div", "td", "th", "li", "section", "tr", "tbody",
                                "thead", "h1", "h2", "h3", "h4", "h5", "h6"):
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            elif child.name in ("span", "font", "b", "i", "strong", "em", "a",
                                "u", "sub", "sup", "small", "label"):
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            else:
                text = child.get_text(strip=True)
                if text:
                    parts.append(text)
        if is_inline:
            result = " ".join(p for p in parts if p.strip())
            result = re.sub(r" {2,}", " ", result)
        else:
            result = "\n\n".join(p for p in parts if p.strip())
        return result.strip()

    result = extract_content(soup)

    # 后处理：中文章节标题分割
    section_pats = [
        r'(?<!\n)\s+(?=[一二三四五六七八九十]+[、．])',
        r'(?<!\n)\s+(?=[（][一二三四五六七八九十][）])',
        r'(?<=[。！？])\s+(?=\d+[.、]\s*[\u4e00-\u9fff])',
        r'(?<!\n)\s{2,}(?=[一二三四五六七八九十]+[、．])',
    ]
    for pat in section_pats:
        result = re.sub(pat, '\n\n', result)

    return result.strip()


def fetch_list(limit=0):
    """获取列表页，返回items[{title, date, url}]"""
    url = f"{BASE_URL}{LIST_PATH}"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"列表页请求失败: {e}")
        return []

    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="news-ul")
    if not ul:
        log("未找到news-ul列表")
        return []

    items = []
    for li in ul.find_all("li"):
        a = li.find("a")
        tt = li.find("div", class_="news-ul-tt")
        dt = li.find("div", class_="news-ul-date")
        if a and tt:
            href = a.get("href", "").strip()
            if href and not href.startswith("http"):
                # wap page path: ./202607/t20260720_245738_wap.html
                if href.startswith("."):
                    href = href.lstrip(".")
                href = BASE_URL + DETAIL_DIR + "/" + href.lstrip("/")
            title = clean_title(tt.get_text(strip=True))
            date_str = dt.get_text(strip=True) if dt else ""
            items.append({"title": title, "date": date_str, "url": href})

    if limit > 0:
        items = items[:limit]

    return items


def fetch_detail(url):
    """获取详情页，返回(title, pub_date, content_html)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"详情页请求失败 {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # 标题
    title = None
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        h1 = soup.find("h1", class_="article-tt")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).split("-")[0].strip()
    title = clean_title(title) if title else ""

    # 日期
    pub_date = None
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]
    if not pub_date:
        date_div = soup.find("div", class_="article-date")
        if date_div:
            txt = date_div.get_text(strip=True)
            m = re.search(r"(\d{4})年(\d{1,2})月(\d{1,2})日", txt)
            if m:
                pub_date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"

    # 正文 - TRS_Editor
    content_html = ""
    trs_editor = soup.find("div", class_="TRS_Editor")
    if trs_editor:
        content_html = trs_editor.encode_contents().decode("utf-8")
    else:
        # fallback: article-p
        art_p = soup.find("div", class_="article-p")
        if art_p:
            content_html = art_p.encode_contents().decode("utf-8")

    return title, pub_date, content_html


def push_to_searchdb(items, source_name="ynsthjt_npzx", group_name="云南省",
                     industry="环评公示"):
    """入库到 search.db"""
    if not items:
        log("无数据可入库")
        return

    db_path = "/mnt/data/search.db"
    import subprocess

    inserted = 0
    for item in items:
        title = item.get("title", "")
        date = item.get("date", "")
        url = item.get("url", "")
        content = item.get("content", "")

        if not title or not url:
            continue

        esc_title = title.replace("'", "''")
        esc_url = url.replace("'", "''")
        esc_source = source_name.replace("'", "''")
        esc_group = group_name.replace("'", "''")
        esc_industry = industry.replace("'", "''")
        esc_date = date.replace("'", "''")
        esc_content = content.replace("'", "''") if content else ""

        use_sql = (
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, source_url, page_url, title, publish_date, summary, content, group_name, industry) "
            "VALUES ("
            f"'{esc_source}', "
            f"'{esc_url}', "
            f"'{esc_url}', "
            f"'{esc_title}', "
            f"'{esc_date}', "
            f"'{esc_title}', "
            f"'{esc_content}', "
            f"'{esc_group}', "
            f"'{esc_industry}'"
            ");"
        )

        cmd = ["sqlite3", db_path, f"PRAGMA busy_timeout=15000; {use_sql}"]
        try:
            result = subprocess.run(
                cmd, capture_output=True, text=True, timeout=30
            )
            if result.returncode == 0:
                inserted += 1
        except Exception:
            pass

    log(f"入库完成: {inserted}/{len(items)} 条")

    # FTS同步
    log("同步FTS...")
    esc_source_safe = source_name.replace("'", "''")
    fts_sql = (
        "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r "
        f"WHERE r.site_name='{esc_source_safe}' "
        "AND NOT EXISTS (SELECT 1 FROM gov_search s WHERE s.rowid=r.id);"
    )
    try:
        subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", db_path, fts_sql],
            capture_output=True, text=True, timeout=30
        )
    except Exception as e:
        log(f"FTS同步失败: {e}")


def main():
    import argparse
    parser = argparse.ArgumentParser(description="云南省生态环境厅-拟审查项目公示爬虫")
    parser.add_argument("--pages", type=int, default=DEFAULT_PAGES,
                        help="限制最多爬取条目数（实际条数=pages×10，默认全部240条）")
    parser.add_argument("--dry-run", action="store_true",
                        help="仅预览不实际爬取详情")
    args = parser.parse_args()

    limit = args.pages * 10 if args.pages and args.pages > 0 else 0
    log(f"🚀 爬虫启动: ynsthjt_npzx, limit={limit or 'all'}, dry_run={args.dry_run}")

    all_items = fetch_list(limit=limit)
    log(f"列表获取到 {len(all_items)} 条")

    if args.dry_run:
        for item in all_items[:5]:
            print(f"  [{item['date']}] {item['title'][:50]}")
            print(f"  URL: {item['url']}")
        log("✅ 预览完成")
        return

    for i, item in enumerate(all_items):
        log(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}...")
        detail_title, detail_date, content_html = fetch_detail(item["url"])
        final_title = detail_title or item["title"]
        final_date = detail_date or item["date"]
        cleaned_content = clean_html(content_html) if content_html else ""

        item["title"] = final_title
        item["date"] = final_date
        item["content"] = cleaned_content
        time.sleep(0.5)

    log(f"\n=== 爬取完成: {len(all_items)} 条 ===")
    for item in all_items[:3]:
        print(f"  [{item['date']}] {item['title'][:50]}")
        print(f"  URL: {item['url']}")
        print(f"  正文: {len(item.get('content', ''))} chars")
        print()

    if all_items:
        push_to_searchdb(all_items)

    log("✅ 全部完成")


if __name__ == "__main__":
    main()
