#!/usr/bin/env python3
"""
Crawler: 梅州市兴宁市叶塘镇人民政府 - 其他 (gkmlpt)
API: Lonsun gkmlpt platform
"""
import sys, os, json, re, time, subprocess
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE = "兴宁市叶塘镇"
COLUMN = "其他"
PROVINCE = "广东"
CATEGORY_ID = 8249
BASE_URL = "https://www.xingning.gov.cn"
APP_BASE = "http://www.xingning.gov.cn/mzxnytzzf"
SID = "753152"
TOTAL_PAGES = 2   # need 2 API pages for up to 200 items (39 items, only page 0 needed)

DEFAULT_FULL_PAGES = 1   # 1 API page = enough for 39 items
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

ATTACH_EXTS = ('.doc', '.docx', '.pdf', '.xls', '.xlsx', '.ppt', '.pptx',
               '.zip', '.rar', '.7z', '.tar', '.gz', '.txt',
               'downfile.jsp', 'download.jsp', 'downfile', 'download')


def log(msg):
    print(f"[{SITE}] {msg}", flush=True)


def fetch_json(url):
    """Fetch API endpoint and return parsed JSON."""
    try:
        r = subprocess.run(
            ["curl", "-sL", "--max-time", "15", "-A", "Mozilla/5.0", url],
            capture_output=True, text=True, timeout=20
        )
        return json.loads(r.stdout)
    except Exception as e:
        log(f"API error: {e}")
        return None


def fetch_html(url):
    """Fetch a detail page and extract DETAIL.content from the JS config."""
    try:
        r = subprocess.run(
            ["curl", "-sL", "--max-time", "15", "-A", "Mozilla/5.0", url],
            capture_output=True, text=True, timeout=20
        )
        html = r.stdout
    except Exception as e:
        log(f"Fetch error for {url}: {e}")
        return None

    # Extract DETAIL: embedded JSON object from window._CONFIG
    idx = html.find("DETAIL:")
    if idx < 0:
        return None

    obj_start = html.find("{", idx + 7)
    if obj_start < 0:
        return None

    depth = 0
    for i in range(obj_start, len(html)):
        if html[i] == "{":
            depth += 1
        elif html[i] == "}":
            depth -= 1
            if depth == 0:
                detail_str = html[obj_start:i+1]
                try:
                    return json.loads(detail_str)
                except json.JSONDecodeError:
                    return None
    return None


def extract_content(detail, page_url, title):
    """Extract clean text and attachments from DETAIL."""
    content_html = detail.get("content", "") or ""
    if not content_html.strip():
        return "", ""

    soup = BeautifulSoup(content_html, "html.parser")
    attachments = []

    # Extract attachment links BEFORE stripping
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        full_url = href if href.startswith("http") else BASE_URL + href
        if any(ext in href.lower() for ext in ATTACH_EXTS):
            md_link = f"[{text}]({full_url})"
            a_tag.replace_with(md_link)
            attachments.append(f'[{text}]({full_url})')
        else:
            md_link = f"[{text}]({full_url})"
            a_tag.replace_with(md_link)

    # Clean and extract text
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()

    # Extract paragraphs
    paragraphs = []
    for p in soup.find_all('p'):
        text = p.get_text(separator='', strip=True)
        if text:
            paragraphs.append(text)

    content = '\n\n'.join(paragraphs) if paragraphs else content_html.strip()
    att_str = ' | '.join(attachments) if attachments else ""
    if att_str:
        if content:
            content += '\n\n' + att_str
        else:
            content = att_str

    return content, att_str


def crawl_pages(start_page, end_page, cutoff):
    """Crawl a range of API pages, dedup by URL."""
    all_items = []
    seen_urls = set()
    cutoff_dt = datetime.strptime(cutoff, "%Y-%m-%d") if cutoff else None

    for pg in range(start_page, end_page + 1):
        url = f"{APP_BASE}/gkmlpt/api/all/{CATEGORY_ID}?page={pg}&sid={SID}"
        data = fetch_json(url)
        if not data:
            log(f"  Page {pg}: no data, stopping")
            break

        articles = data.get("articles", [])
        if not articles:
            log(f"  Page {pg}: no articles, stopping")
            break

        new_count = 0
        for a in articles:
            page_url = a.get("url", "")
            if not page_url:
                continue
            if page_url.startswith("/"):
                page_url = BASE_URL + page_url

            # Dedup
            if page_url in seen_urls:
                continue
            seen_urls.add(page_url)

            # Date filtering
            ts = a.get("date", 0)
            if ts:
                dt = datetime.fromtimestamp(ts)
                date_str = dt.strftime("%Y-%m-%d")
            else:
                date_str = ""

            if cutoff_dt and date_str:
                try:
                    d = datetime.strptime(date_str, "%Y-%m-%d")
                    if d < cutoff_dt:
                        continue
                except ValueError:
                    pass

            title = (a.get("title", "") or "").strip()
            if not title or len(title) < 3:
                continue

            all_items.append({
                "title": title,
                "url": page_url,
                "date": date_str,
                "ts": ts,
            })
            new_count += 1

        log(f"  Page {pg}: +{new_count} new (total {len(all_items)})")
        if new_count == 0 and pg > start_page:
            break
        time.sleep(0.5)

    return all_items


def process_details(all_items):
    """Fetch detail pages and extract content."""
    log(f"\nFetching details for {len(all_items)} items...")
    enriched = []
    for idx, item in enumerate(all_items, 1):
        detail = fetch_html(item["url"])
        if detail:
            content, attachments = extract_content(detail, item["url"], item["title"])
        else:
            content = ""
            attachments = ""

        enriched.append({
            "title": item["title"],
            "page_url": item["url"],
            "publish_date": item["date"],
            "content": content,
            "attachments": attachments,
            "site_name": f"{SITE}-{COLUMN}",
            "column": COLUMN,
        })

        if idx % 10 == 0:
            log(f"  Progress: {idx}/{len(all_items)}")
        time.sleep(0.3)

    return enriched


def store_items(items):
    """Insert items into search.db."""
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()

    new_count = 0
    skip_count = 0
    for item in items:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (
                    item["site_name"],
                    item["page_url"],
                    item["page_url"],
                    item["title"],
                    item["publish_date"],
                    item["content"],
                    (item["content"] or "")[:500],
                    item["column"],
                )
            )
            if c.rowcount > 0:
                new_count += 1
            else:
                skip_count += 1
        except Exception as e:
            log(f"  DB error: {e}")
            skip_count += 1

    conn.commit()
    conn.close()
    return new_count, skip_count


def crawl_all(months_back=36):
    """New site initial: crawl API pages for initial data."""
    cutoff = (datetime.now() - timedelta(days=months_back * 30)).strftime("%Y-%m-%d")
    log(f"Strategy: initial crawl (first page)")
    items = crawl_pages(0, 0, cutoff)  # Only page 0 since total < 100
    enriched = process_details(items)
    new, skip = store_items(enriched)
    log(f"Result: {new} new, {skip} skipped")


def crawl_incremental():
    """Daily: fetch page 0, filter recent items, fetch details."""
    cutoff = (datetime.now() - timedelta(days=7)).strftime("%Y-%m-%d")
    log(f"Incremental: last 7 days")
    items = crawl_pages(0, 0, cutoff)
    enriched = process_details(items)
    new, skip = store_items(enriched)
    log(f"Incremental result: {new} new, {skip} skipped")


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        arg2 = sys.argv[2] if len(sys.argv) > 2 else ""
        if arg2 == "all":
            crawl_all(months_back=120)
        else:
            months = int(arg2) if arg2 and arg2.isdigit() else 36
            crawl_all(months)
    elif mode == "list":
        data = fetch_json(f"{APP_BASE}/gkmlpt/api/all/{CATEGORY_ID}?page=0&sid={SID}")
        articles = data.get("articles", []) if data else []
        log(f"Total: {len(articles)} articles")
        for a in articles[:5]:
            dt = datetime.fromtimestamp(a.get("date", 0)).strftime("%Y-%m-%d")
            print(f"  {dt} | {a.get('title','')[:60]}")
