#!/usr/bin/env python3
"""
crawl_linyou_jcygk.py - 麟游县人民政府-决策预公开
====================================================
站点: http://www.linyou.gov.cn/col14518/col14521/col14548/
CMS:  麟游WCM (中国政务CMS)
列表: ul > li > a[title]+b(日期)
分页: index.html, index_1.html, index_2.html (3页, 共58条)
详情: <div id="Zoom"> 正文, <meta ArticleTitle>标题, <meta PubDate>日期

用法:
  python3 crawl_linyou_jcygk.py              # 默认5页
  python3 crawl_linyou_jcygk.py --pages 3    # 爬3页
  python3 crawl_linyou_jcygk.py --dry-run    # 仅预览
"""

import sys, os, re, json, requests, time
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "http://www.linyou.gov.cn"
LIST_PATH = "/col14518/col14521/col14548"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

DEFAULT_PAGES = 5


def log(msg):
    ts = datetime.now().strftime("%H:%M:%S")
    print(f"[{ts}] {msg}", flush=True)


def clean_title(title):
    """清洗标题：strip &middot; &nbsp; 等HTML实体前缀"""
    if not title:
        return ""
    title = title.replace("&middot;", "·").replace("&nbsp;", " ").replace("&amp;", "&")
    return title.strip("·. \\t\\n\\r")


def clean_html(content_html):
    """清洗正文HTML：去冗余标签，保留表格HTML"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")

    for tag in soup.find_all(["style", "script"]):
        tag.decompose()
    for br in soup.find_all("br"):
        br.replace_with("\n")
    for tag in soup.find_all(["o:p"]):
        tag.decompose()

    # 修复空<p>嵌套 以及 包裹型<div>嵌套(只含<p>子元素)
    # 如: <div class="trs_editor_view"><p>...</p><p>...</p></div> → 直接解嵌套
    for tag in list(soup.find_all(["p", "div"])):
        kids = [c for c in tag.children if not (c.name is None and not str(c).strip())]
        if kids and all(c.name == "p" for c in kids):
            tag.unwrap()

    def extract_content(element, is_inline=False):
        parts = []
        for child in element.children:
            if child.name is None:
                text = str(child).strip()
                if text:
                    parts.append(text)
            elif child.name == "table":
                parts.append(str(child))
            elif child.name in ("p", "div", "td", "th", "li", "section", "tr", "tbody",
                                "thead", "h1", "h2", "h3", "h4", "h5", "h6"):
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            elif child.name in ("span", "font", "b", "i", "strong", "em", "a",
                                "u", "sub", "sup", "small", "label"):
                inner = extract_content(child, is_inline=True)
                if inner.strip():
                    parts.append(inner)
            else:
                text = child.get_text(strip=True)
                if text:
                    parts.append(text)
        if is_inline:
            result = " ".join(p for p in parts if p.strip())
            result = re.sub(r" {2,}", " ", result)
        else:
            result = "\n\n".join(p for p in parts if p.strip())
        return result.strip()

    result = extract_content(soup)

    # 后处理：中文章节标题分割
    section_pats = [
        r'(?<!\n)\s+(?=[一二三四五六七八九十]+[、．])',
        r'(?<!\n)\s+(?=[（][一二三四五六七八九十][）])',
        r'(?<=[。！？])\s+(?=\d+[.、]\s*[\u4e00-\u9fff])',
        r'(?<!\n)\s{2,}(?=[一二三四五六七八九十]+[、．])',
    ]
    for pat in section_pats:
        result = re.sub(pat, '\n\n', result)

    return result.strip()


def fetch_list_page(page=1):
    """获取单页列表"""
    if page == 1:
        url = f"{BASE_URL}{LIST_PATH}/index.html"
    else:
        url = f"{BASE_URL}{LIST_PATH}/index_{page-1}.html"

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"列表页请求失败 (page={page}): {e}")
        return []

    soup = BeautifulSoup(html, "html.parser")
    items = []

    for ul in soup.find_all("ul"):
        for li in ul.find_all("li", recursive=False):
            a = li.find("a")
            b = li.find("b")
            if a and b:
                href = a.get("href", "").strip()
                if not href:
                    continue
                # Make absolute URL
                if not href.startswith("http"):
                    if href.startswith("./"):
                        href = href[1:]
                    href = BASE_URL + LIST_PATH + "/" + href.lstrip("/")
                title = a.get("title", "") or a.get_text(strip=True)
                title = clean_title(title)
                date = b.get_text(strip=True).strip()
                items.append({"title": title, "date": date, "url": href})

    return items


def fetch_detail(url):
    """获取详情页，返回(title, pub_date, content_html)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text
    except Exception as e:
        log(f"详情页请求失败 {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # 标题
    title = None
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).split(" ")[0].strip()
    title = clean_title(title) if title else ""

    # 日期
    pub_date = None
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # 正文 - Zoom
    content_html = ""
    zoom_div = soup.find("div", id="Zoom")
    if zoom_div:
        content_html = zoom_div.encode_contents().decode("utf-8")
    else:
        # fallback: try other common containers
        for cls in ["TRS_Editor", "Custom_UnionStyle", "con_text", "info-content"]:
            fb = soup.find("div", class_=cls)
            if fb:
                content_html = fb.encode_contents().decode("utf-8")
                break

    return title, pub_date, content_html


def push_to_searchdb(items, source_name="linyou_jcygk", group_name="陕西省",
                     industry="政府公告"):
    """入库到 search.db"""
    if not items:
        log("无数据可入库")
        return

    db_path = "/mnt/data/search.db"
    import subprocess

    inserted = 0
    for item in items:
        title = item.get("title", "")
        date = item.get("date", "")
        url = item.get("url", "")
        content = item.get("content", "")

        if not title or not url:
            continue

        esc_title = title.replace("'", "''")
        esc_url = url.replace("'", "''")
        esc_source = source_name.replace("'", "''")
        esc_group = group_name.replace("'", "''")
        esc_industry = industry.replace("'", "''")
        esc_date = date.replace("'", "''")
        esc_content = content.replace("'", "''") if content else ""

        use_sql = (
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, source_url, page_url, title, publish_date, summary, content, group_name, industry) "
            "VALUES ("
            f"'{esc_source}', "
            f"'{esc_url}', "
            f"'{esc_url}', "
            f"'{esc_title}', "
            f"'{esc_date}', "
            f"'{esc_title}', "
            f"'{esc_content}', "
            f"'{esc_group}', "
            f"'{esc_industry}'"
            ");"
        )

        cmd = ["sqlite3", db_path, f"PRAGMA busy_timeout=15000; {use_sql}"]
        try:
            result = subprocess.run(
                cmd, capture_output=True, text=True, timeout=30
            )
            if result.returncode == 0:
                inserted += 1
        except Exception:
            pass

    log(f"入库完成: {inserted}/{len(items)} 条")

    # FTS同步
    log("同步FTS...")
    esc_source_safe = source_name.replace("'", "''")
    fts_sql = (
        "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r "
        f"WHERE r.site_name='{esc_source_safe}' "
        "AND NOT EXISTS (SELECT 1 FROM gov_search s WHERE s.rowid=r.id);"
    )
    try:
        subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", db_path, fts_sql],
            capture_output=True, text=True, timeout=30
        )
    except Exception as e:
        log(f"FTS同步失败: {e}")


def main():
    import argparse
    parser = argparse.ArgumentParser(description="麟游县人民政府-决策预公开爬虫")
    parser.add_argument("--pages", type=int, default=DEFAULT_PAGES,
                        help=f"爬取页数 (默认: {DEFAULT_PAGES}, 共3页)")
    parser.add_argument("--dry-run", action="store_true",
                        help="仅预览不实际爬取详情")
    args = parser.parse_args()

    pages = min(args.pages, 3)  # 最多3页
    log(f"🚀 爬虫启动: linyou_jcygk, pages={pages}, dry_run={args.dry_run}")

    all_items = []
    for page in range(1, pages + 1):
        log(f"爬取第 {page} 页...")
        items = fetch_list_page(page)
        if not items:
            log(f"第 {page} 页无数据，停止")
            break
        log(f"  获取到 {len(items)} 条")
        all_items.extend(items)

    log(f"列表总计: {len(all_items)} 条")

    if args.dry_run:
        for item in all_items[:5]:
            print(f"  [{item['date']}] {item['title'][:50]}")
            print(f"  URL: {item['url']}")
        log("✅ 预览完成")
        return

    for i, item in enumerate(all_items):
        log(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}...")
        detail_title, detail_date, content_html = fetch_detail(item["url"])
        final_title = detail_title or item["title"]
        final_date = detail_date or item["date"]
        cleaned_content = clean_html(content_html) if content_html else ""

        item["title"] = final_title
        item["date"] = final_date
        item["content"] = cleaned_content
        time.sleep(0.5)

    log(f"\n=== 爬取完成: {len(all_items)} 条 ===")
    for item in all_items[:3]:
        print(f"  [{item['date']}] {item['title'][:50]}")
        print(f"  URL: {item['url']}")
        print(f"  正文: {len(item.get('content', ''))} chars")
        print()

    if all_items:
        push_to_searchdb(all_items)

    log("✅ 全部完成")


if __name__ == "__main__":
    main()
