#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
永新县人民政府 - 公告公示爬虫
CMS: Dayrui/FineCMS (ZCMS)
列表: POST api-ajax_list-{page}.html (ajax_type[]=jQuery序列化)
详情: news-show-{id}.html
WAF: CloudWAF (Referer绕过)

段落结构: div.conTxt > p > span (每段独立p标签)
附件: 正文中的a链接指向.pdf/.doc/.xls等文件
"""

import os, re, sys, time, json, subprocess
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE_URL = "http://www.yongxin.gov.cn"
MAX_PAGES = 8  # 共227条, 30条/页
SITE_NAME = "永新县人民政府-公告公示"
INDUSTRY = "政府公告"
GROUP = "江西"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": BASE_URL + "/news-list-gonggaogongshi.html",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def fetch(url, retries=3, post_data=None, session=None):
    """通用请求函数，支持GET和POST，自动处理CloudWAF"""
    s = session or requests.Session()
    for i in range(retries):
        try:
            if post_data:
                r = s.post(url, data=post_data, headers=HEADERS, timeout=30)
            else:
                r = s.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None


def build_ajax_params(page):
    """构建jQuery序列化的ajax_type[]参数"""
    # ZCMS API参数格式
    params = [
        ("ajax_type[]", "11_news"),
        ("ajax_type[]", "21"),
        ("ajax_type[]", "11"),
        ("ajax_type[]", "news"),
        ("ajax_type[]", "Y-m-d"),
        ("ajax_type[]", "40"),
        ("ajax_type[]", "30"),
        ("ajax_type[]", "is_top DESC"),
        ("ajax_type[]", "displayorder DESC"),
        ("ajax_type[]", "inputtime DESC"),
        ("ajax_type[]", ""),
        ("is_ds", "1"),
    ]
    return params


def parse_list(html):
    """解析列表页JSON，返回(item列表, total)"""
    try:
        data = json.loads(html)
        items = data.get("data", [])
        total = data.get("total", 0)
        return items, total
    except Exception as e:
        print(f"  解析列表JSON失败: {e}", file=sys.stderr)
        return [], 0


def fetch_detail(url, session):
    """获取详情页并解析"""
    html = fetch(url, session=session)
    if not html:
        return None

    soup = BeautifulSoup(html, "html.parser")
    result = {}

    # 标题
    meta_title = soup.select_one('meta[name="ArticleTitle"]')
    if meta_title and meta_title.get("content"):
        result["title"] = meta_title["content"].strip()
    if not result.get("title"):
        intr = soup.select_one("div.intr_title")
        if intr:
            result["title"] = intr.get_text(strip=True)

    # 发布日期
    meta_pub = soup.select_one('meta[name="PubDate"]')
    if meta_pub and meta_pub.get("content"):
        result["publish_date"] = meta_pub["content"].strip()[:10]

    # 来源
    meta_src = soup.select_one('meta[name="ContentSource"]')
    if meta_src and meta_src.get("content"):
        result["source"] = meta_src["content"].strip()

    # 正文 - 分段处理
    content_div = soup.select_one("div.conTxt")
    paragraphs = []
    has_table = False
    attachments = []

    if content_div:
        for child in content_div.children:
            if child.name == "p":
                if child.find_parent("table"):
                    continue
                # 检查附件链接
                for link in child.find_all("a"):
                    href = link.get("href", "")
                    if href and re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", href, re.I):
                        attach_text = link.get_text(strip=True) or "附件"
                        if href.startswith("/"):
                            href = BASE_URL + href
                        elif not href.startswith("http"):
                            href = BASE_URL + "/" + href
                        attachments.append(f"[{attach_text}]({href})")

                text = child.get_text(strip=True)
                if text:
                    if child.find("table"):
                        has_table = True
                    paragraphs.append(text)

            elif child.name == "table":
                has_table = True
                paragraphs.append(str(child))

        if not paragraphs:
            text = content_div.get_text(strip=True)
            if text:
                paragraphs.append(text)

    result["content"] = "\n\n".join(paragraphs) if paragraphs else ""
    result["has_table"] = has_table
    result["attachments"] = "; ".join(attachments) if attachments else ""
    return result


def esc(s):
    """SQLite转义单引号"""
    return s.replace("'", "''")


def run_sql(sql):
    """通过stdin管道执行SQL（避免Argument list too long）"""
    proc = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH],
        input=sql, capture_output=True, text=True, timeout=10
    )
    if proc.returncode != 0:
        raise RuntimeError(f"SQLite error: {proc.stderr.strip()}")
    return proc.stdout.strip()


def save_to_db(page_url, title, content, publish_date, site_name,
               group_name, industry, source="", attachments="", has_table=0):
    """插入到search.db（stdin管道防Argument list too long）"""
    # 检查是否已存在
    check_sql = f"SELECT rowid FROM gov_raw WHERE page_url = '{esc(page_url)}';\n"
    result = run_sql(check_sql)
    if result:
        return False  # 已存在

    # 摘要（取前200字符）
    summary = re.sub(r'\s+', ' ', content[:200]).strip() if content else ""

    sql = f"""INSERT INTO gov_raw 
        (page_url, title, content, publish_date, site_name, 
         group_name, industry, source_url, summary, attachments, has_table)
    VALUES (
        '{esc(page_url)}',
        '{esc(title)}',
        '{esc(content)}',
        '{esc(publish_date)}',
        '{esc(site_name)}',
        '{esc(group_name)}',
        '{esc(industry)}',
        '{esc(page_url)}',
        '{esc(summary)}',
        '{esc(attachments)}',
        {int(has_table)}
    );
"""
    try:
        run_sql(sql)
    except RuntimeError as e:
        print(f"  入库失败: {e}", file=sys.stderr)
        return False

    # FTS同步
    sync_sql = f"""INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
SELECT rowid, '{esc(title)}', '{esc(site_name)}', '{esc(summary)}'
FROM gov_raw WHERE page_url = '{esc(page_url)}'
AND rowid NOT IN (SELECT rowid FROM gov_search WHERE title = '{esc(title)}');
"""
    try:
        run_sql(sync_sql)
    except RuntimeError as e:
        print(f"  FTS同步失败: {e}", file=sys.stderr)
    return True


def crawl(max_pages=5):
    """主爬虫逻辑"""
    session = requests.Session()
    session.headers.update(HEADERS)

    # 先访问首页获取CloudWAF Cookie
    fetch(f"{BASE_URL}/", session=session)
    time.sleep(0.5)

    total_inserted = 0
    total_skipped = 0
    page = 1

    while page <= max_pages:
        print(f"--- 第{page}页 ---")

        params = build_ajax_params(page)
        list_url = f"{BASE_URL}/api-ajax_list-{page}.html"
        html = fetch(list_url, post_data=params, session=session)

        if not html:
            page += 1
            continue

        items, total = parse_list(html)
        if not items:
            print(f"  第{page}页无数据，停止")
            break

        print(f"  获取{len(items)}条 (共{total}条)")

        for i, item in enumerate(items):
            title = (item.get("title") or "").strip()
            url = (item.get("url") or "").strip()
            list_date = (item.get("inputtime") or "")[:10]
            source = (item.get("laiyuan") or "").strip()

            if not title or not url:
                total_skipped += 1
                continue

            print(f"  [{i+1}] {title[:50]}...", end="", flush=True)

            # 获取详情
            detail = fetch_detail(url, session)

            if detail:
                final_title = detail.get("title") or title
                final_date = detail.get("publish_date") or list_date
                final_source = detail.get("source") or source
                content = detail.get("content", "")
                has_table = detail.get("has_table", False)
                attachments = detail.get("attachments", "")

                result = save_to_db(
                    page_url=url,
                    title=final_title,
                    content=content,
                    publish_date=final_date,
                    site_name=SITE_NAME,
                    group_name=GROUP,
                    industry=INDUSTRY,
                    source=final_source,
                    attachments=attachments,
                    has_table=1 if has_table else 0,
                )

                if result:
                    total_inserted += 1
                    print(f" ✓ (含表格)" if has_table else " ✓")
                else:
                    total_skipped += 1
                    print(f" 已存在")
            else:
                total_skipped += 1
                print(f" 详情获取失败")

            time.sleep(0.3)

        page += 1

    print(f"\n✅ 爬取完成: 共爬取{page-1}页, 新增{total_inserted}条, 跳过{total_skipped}条")
    return total_inserted


def main():
    import argparse
    parser = argparse.ArgumentParser(description="永新县人民政府-公告公示爬虫")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数(默认5)")
    args = parser.parse_args()

    print(f"永新县政府-公告公示爬虫 (pages={args.pages})")
    print(f"分组: {GROUP}, 行业: {INDUSTRY}")
    t0 = time.time()
    cnt = crawl(max_pages=args.pages)
    print(f"耗时: {time.time()-t0:.1f}s, 新增: {cnt}条")


if __name__ == "__main__":
    main()
