#!/usr/bin/env python3
"""
福鼎市-通知公告 (www.fuding.gov.cn)
==============================
CMS: TRS (TRSWAS5 API)
API: /was5/web/search
数据直接用 API 返回（含完整 content），不需逐条抓详情页

用法:
    python3 crawl_fuding.py          # 全量(最多5页)
    python3 crawl_fuding.py --test   # 测试 5 条
"""

import re, sys, os, json, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
import os
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME  = "福鼎市-通知公告"
API_URL    = "http://www.fuding.gov.cn/was5/web/search"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES  = 5
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

def fetch_page(page):
    """调用 WAS5 API 获取列表页数据"""
    params = {
        "channelid": "238418",
        "perpage": "10",
        "page": str(page),
        "searchword": "",
        "orderby": "-docreltime",
        "classsql": "chnlid=13725",
    }
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        # TRS API returns \r\n inside content, use strict=False
        data = json.loads(r.text, strict=False)
        return data
    except Exception as e:
        print("  ! API fail page " + str(page) + ": " + str(e)[:80])
        return None

def parse_api_response(data):
    """从 API 响应中提取条目"""
    items = []
    docs = data.get("docs", [])
    for doc in docs:
        title = doc.get("title", "").strip()
        # Skip TRS header row
        if title in ["文章标题", "标题", "doc_title", "", "发布时间"]:
            continue
        url = doc.get("url", "")
        pub_date = doc.get("time", "")[:10]
        content = doc.get("html", "")
        # Clean content
        if content:
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
            content = content.strip()
        items.append({
            "title": title,
            "url": url,
            "content": content,
            "pub_date": pub_date,
        })
    return items

def to_db(items):
    if not items:
        print("  - no data")
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR REPLACE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, tags, script_name) VALUES (?,?,?,?,?,?,?, 'crawl_fuding.py')",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "通知公告",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
    #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
    db.commit()
    db.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    db.commit()
    db.close()
    print("  [DB] new: " + str(ok) + " skip: " + str(fail) + " FTS synced")
    return ok, fail

def crawl(test=False):
    all_items = []
    seen_urls = set()

    for page in range(1, MAX_PAGES + 1):
        data = fetch_page(page)
        if not data or not data.get("docs"):
            print("  - page " + str(page) + ": no data, reached end")
            break
        items = parse_api_response(data)
        new = 0
        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
            new += 1
        f = items[0]["pub_date"] if items else "?"
        l = items[-1]["pub_date"] if items else "?"
        total = data.get("count", "?")
        print("  [Page " + str(page) + "] " + str(len(items)) + " items (" + f + " ~ " + l + "), new: " + str(new) + " (total: " + str(total) + ")")

    if test:
        all_items = all_items[:5]
        print("  TEST mode: " + str(len(all_items)) + " items")

    print("  Total: " + str(len(all_items)) + " items (within 3yr)")
    if not all_items:
        return 0, 0

    ok, fail = to_db(all_items)
    return ok, fail

if __name__ == "__main__":
    test = "--test" in sys.argv
    mode = "TEST" if test else "FULL"
    print()
    print("[" + SITE_NAME + "] " + mode + " (max " + str(MAX_PAGES) + " pages, 3yr)")
    print("  API: " + API_URL)
    t0 = time.time()
    ok, fail = crawl(test=test)
    print("  Time: " + str(round(time.time()-t0, 1)) + "s")
    print("  New: " + str(ok) + " Skip: " + str(fail))
