#!/usr/bin/env python3
"""
Crawler: 丰城市人民政府 - 回应关切 (UCAP CMS, JSON API)
API: POST /queryList JSON body {siteCode, channelCode, pageSize, current}
Returns full content directly - no detail page fetch needed.
"""
import sys, os, json, re, time, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE = "丰城市人民政府"
COLUMN = "回应关切"
BASE_URL = "https://www.jxfc.gov.cn"
API_URL = f"{BASE_URL}/queryList"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/json;charset=UTF-8",
}
PAGE_SIZE = 15


def log(msg):
    print(f"[{SITE}] {msg}", flush=True)


def fetch_page(page_no):
    body = json.dumps({
        "pageSize": PAGE_SIZE,
        "current": page_no,
        "siteCode": "fcsrmzf",
        "channelCode": "hygq2c"
    })
    try:
        r = requests.post(API_URL, data=body, headers=HEADERS, timeout=30, verify=False)
        return r.json()
    except Exception as e:
        log(f"API error page {page_no}: {e}")
        return None


def extract_clean_text(content_html):
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        full_url = urljoin(BASE_URL, href)
        a_tag.replace_with(f"[{text}]({full_url})")
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()
    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            t = el.get_text(separator='', strip=True)
            if t:
                parts.append(t)
    return '\n\n'.join(parts) if parts else content_html.strip()


def crawl_all(months_back=36):
    cutoff = (datetime.now() - timedelta(days=months_back * 30)).strftime("%Y-%m-%d")
    log(f"Full crawl (cutoff: {cutoff})")

    first = fetch_page(1)
    if not first:
        return
    total = first["data"]["total"]
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    log(f"Total: {total}, pages: {total_pages}")

    enriched = []
    for pg in range(1, total_pages + 1):
        data = fetch_page(pg)
        if not data:
            break
        for item in data["data"]["results"]:
            src = item["source"]
            pub_date = (src.get("pubDate", "") or "")[:10]
            if pub_date < cutoff:
                continue
            title = src.get("title", "")
            content = src.get("content", {}).get("content", "")
            clean = extract_clean_text(content)

            enriched.append({
                "title": title,
                "page_url": src.get("urls", "{}").replace('"', '').split(":")[-1].strip().strip("}") if
                            isinstance(src.get("urls"), str) and "pc" in src.get("urls", "") else "",
                "publish_date": pub_date,
                "content": clean,
                "site_name": f"{SITE}-{COLUMN}",
                "column": COLUMN,
            })
        log(f"  Page {pg}: processed")
        time.sleep(0.5)

    log(f"After filter: {len(enriched)} items")

    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"]))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


def crawl_incremental():
    cutoff = (datetime.now() - timedelta(days=7)).strftime("%Y-%m-%d")
    log(f"Incremental (since {cutoff})")
    data = fetch_page(1)
    if not data:
        return

    enriched = []
    for item in data["data"]["results"]:
        src = item["source"]
        pub_date = (src.get("pubDate", "") or "")[:10]
        if pub_date < cutoff:
            continue
        title = src.get("title", "")
        content = src.get("content", {}).get("content", "")
        clean = extract_clean_text(content)

        enriched.append({
            "title": title,
            "page_url": "",
            "publish_date": pub_date,
            "content": clean,
            "site_name": f"{SITE}-{COLUMN}",
            "column": COLUMN,
        })
        time.sleep(0.3)

    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"]))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        crawl_all()
    elif mode == "list":
        data = fetch_page(1)
        if data:
            log(f"Total: {data['data']['total']}, first 5 items:")
            for item in data["data"]["results"][:5]:
                src = item["source"]
                print(f"  {(src.get('pubDate',''))[:10]} | {src.get('title','')[:60]}")
