#!/usr/bin/env python3
import os
"""
crawl_tonggu.py — 铜鼓县人民政府·重大建设项目
============================================
Endpoint: POST /searchManuscript (数融CMS)
Channel: 重大建设项目 (yjzjjfk5fvt)
Total: 41条 (近3年)
API返回完整正文，无需二次请求。

用法:
    python3 crawl_tonggu.py             # 全量爬
    python3 crawl_tonggu.py 1           # 增量（只爬1页）
"""

import json, os, sys, re, time, requests

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "铜鼓县重大建设项目"
BASE_URL = "http://www.tonggu.gov.cn"
API_URL = f"{BASE_URL}/searchManuscript"
CHANNEL_ID = "2014209936484589568"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Content-Type": "application/json",
    "Referer": f"{BASE_URL}/tgxrmzf/yjzjjfk5fvt/pc/list.html",
}
THREE_YEARS = time.time() - 3 * 365 * 24 * 3600


def norm_date(d):
    """标准化日期为 YYYY-MM-DD"""
    if not d:
        return ""
    m = re.match(r"(\d{4})-(\d{1,2})-(\d{1,2})", str(d))
    if m:
        return f"{int(m.group(1)):04d}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return str(d)[:10]


def clean_content(html):
    """清理正文HTML，移除mce-style等编辑痕迹但保留结构标签"""
    if not html:
        return ""
    # 移除 data-mce-* 属性
    html = re.sub(r'\sdata-mce-[a-zA-Z-]+="[^"]*"', "", html)
    # 移除 mce-style 但不移除 style
    return html.strip()


def crawl_list(incremental=False):
    """爬取列表，返回 {title, url, pub_date, content, source} 列表"""
    all_items = []
    page = 1
    max_pages = 1 if incremental else 5  # 首页 or 5页

    while page <= max_pages:
        payload = {
            "channelTreeIds": [CHANNEL_ID],
            "current": page,
            "pageSize": 50,
            "title": "",
            "Contenthtml": "",
            "isSearchHighlight": False,
        }
        try:
            resp = requests.post(API_URL, json=payload, headers=HEADERS, timeout=30)
            resp.raise_for_status()
            data = resp.json()
        except Exception as e:
            print(f"  ❌ 第{page}页请求失败: {e}")
            break

        results = data.get("data", {}).get("results", [])
        if not results:
            print(f"  📄 第{page}页: 无数据，结束")
            break

        total = data["data"].get("total", 0)
        print(f"  📄 第{page}页: {len(results)}条 (共{total}条)", end="")

        new_count = 0
        for item in results:
            title = re.sub(r'<[^>]+>', '', (item.get("showTitle") or "").strip())
            pub_date = norm_date(item.get("pubDate", ""))
            content_source = item.get("contentSource") or ""
            raw_content = item.get("content", {}).get("content", "")
            content = clean_content(raw_content)

            # 日期过滤：近3年
            if pub_date and pub_date < "2023-06":
                continue

            # 从 urls 字段取连接
            urls_raw = item.get("urls", "{}")
            try:
                urls = json.loads(urls_raw) if isinstance(urls_raw, str) else urls_raw
            except json.JSONDecodeError:
                urls = {}
            detail_url = urls.get("pc", "")
            if detail_url and not detail_url.startswith("http"):
                detail_url = f"{BASE_URL}{detail_url}"

            all_items.append({
                "title": title,
                "url": detail_url,
                "pub_date": pub_date,
                "content": content,
                "source": content_source or SITE_NAME,
                "site_name": SITE_NAME,
            })
            new_count += 1

        print(f" → 通过日期过滤: {new_count}条")
        page += 1

    return all_items


def main():
    incremental = len(sys.argv) > 1 and sys.argv[1] in ("1", "--incremental")
    mode = "增量" if incremental else "全量"
    print(f"🔍 {SITE_NAME} [{mode}]")

    t0 = time.time()
    items = crawl_list(incremental)
    elapsed = time.time() - t0

    if not items:
        print(f"  ⏭ 无数据")
        return

    print(f"\n📊 共获取 {len(items)} 条，耗时 {elapsed:.0f}s")

    # 入库
    import sqlite3

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    ok, skip = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, summary, content, status, category, tags) "
                "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
                (
                    it["site_name"][:200],
                    it["url"],
                    it["url"],
                    (it["title"] or "")[:500],
                    it["pub_date"],
                    it["title"][:500],  # summary = title
                    it["content"],
                    "active",
                    "重大建设项目",
                    "",
                ),
            )
            if db.total_changes > 0:
                ok += 1
            else:
                skip += 1
        except Exception as e:
            skip += 1
            print(f"  ❌ 入库失败: {e}")

    db.commit()

    # 同步FTS
    db.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary FROM gov_raw r "
        "WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,),
    )
    db.commit()
    db.close()

    print(f"  💾 入库: 新增{ok}, 跳过{skip}")


if __name__ == "__main__":
    main()
