#!/usr/bin/env python3
"""
Crawler: 梅州市生态环境局蕉岭分局 - 其他 (gkmlpt)
API: Lonsun gkmlpt platform
"""
import sys, os, json, re, time, subprocess
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE = "蕉岭县生态环境"
COLUMN = "其他"
PROVINCE = "广东"
CATEGORY_ID = 12199
BASE_URL = "https://www.jiaoling.gov.cn"
APP_BASE = "http://www.jiaoling.gov.cn/mzqlhbj"
SID = "753227"

ATTACH_EXTS = ('.doc', '.docx', '.pdf', '.xls', '.xlsx', '.ppt', '.pptx',
               '.zip', '.rar', '.7z', '.tar', '.gz', '.txt')


def log(msg):
    print(f"[{SITE}] {msg}", flush=True)


def fetch_json(url):
    try:
        r = subprocess.run(
            ["curl", "-sL", "--max-time", "15", "-A", "Mozilla/5.0", url],
            capture_output=True, text=True, timeout=20
        )
        return json.loads(r.stdout)
    except Exception as e:
        log(f"API error: {e}")
        return None


def fetch_html(url):
    try:
        r = subprocess.run(
            ["curl", "-sL", "--max-time", "15", "-A", "Mozilla/5.0", url],
            capture_output=True, text=True, timeout=20
        )
        html = r.stdout
    except Exception as e:
        return None

    # Extract DETAIL JSON from window._CONFIG
    idx = html.find("DETAIL:")
    if idx < 0:
        return None
    obj_start = html.find("{", idx + 7)
    if obj_start < 0:
        return None
    depth = 0
    for i in range(obj_start, len(html)):
        if html[i] == "{":
            depth += 1
        elif html[i] == "}":
            depth -= 1
            if depth == 0:
                detail_str = html[obj_start:i+1]
                try:
                    return json.loads(detail_str)
                except json.JSONDecodeError:
                    return None
    return None


def extract_content(detail):
    content_html = detail.get("content", "") or ""
    if not content_html.strip():
        return ""

    soup = BeautifulSoup(content_html, "html.parser")

    # Attachments and links to markdown
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        full_url = href if href.startswith("http") else BASE_URL + href
        md_link = f"[{text}]({full_url})"
        a_tag.replace_with(md_link)

    # Unwrap inline tags
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()

    paragraphs = []
    for p in soup.find_all('p'):
        text = p.get_text(separator='', strip=True)
        if text:
            paragraphs.append(text)

    return '\n\n'.join(paragraphs) if paragraphs else content_html.strip()


def crawl_pages(start_page, end_page, cutoff):
    all_items = []
    seen_urls = set()
    cutoff_dt = datetime.strptime(cutoff, "%Y-%m-%d") if cutoff else None

    for pg in range(start_page, end_page + 1):
        url = f"{APP_BASE}/gkmlpt/api/all/{CATEGORY_ID}?page={pg}&sid={SID}"
        data = fetch_json(url)
        if not data:
            break

        articles = data.get("articles", [])
        if not articles:
            break

        new_count = 0
        for a in articles:
            page_url = a.get("url", "")
            if not page_url:
                continue
            if page_url.startswith("/"):
                page_url = BASE_URL + page_url

            if page_url in seen_urls:
                continue
            seen_urls.add(page_url)

            ts = a.get("date", 0)
            date_str = datetime.fromtimestamp(ts).strftime("%Y-%m-%d") if ts else ""

            if cutoff_dt and date_str:
                try:
                    d = datetime.strptime(date_str, "%Y-%m-%d")
                    if d < cutoff_dt:
                        continue
                except ValueError:
                    pass

            title = (a.get("title", "") or "").strip()
            if not title or len(title) < 3:
                continue

            all_items.append({"title": title, "url": page_url, "date": date_str})
            new_count += 1

        log(f"  Page {pg}: +{new_count} new (total {len(all_items)})")
        if new_count == 0 and pg > start_page:
            break
        time.sleep(0.5)

    return all_items


def process_details(all_items):
    log(f"Fetching details for {len(all_items)} items...")
    enriched = []
    for idx, item in enumerate(all_items, 1):
        detail = fetch_html(item["url"])
        content = extract_content(detail) if detail else ""

        enriched.append({
            "title": item["title"],
            "page_url": item["url"],
            "publish_date": item["date"],
            "content": content,
            "site_name": f"{SITE}-{COLUMN}",
            "column": COLUMN,
        })

        if idx % 10 == 0:
            log(f"  Progress: {idx}/{len(all_items)}")
        time.sleep(0.3)

    return enriched


def store_items(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()

    new_count = 0
    skip_count = 0
    for item in items:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (
                    item["site_name"],
                    item["page_url"],
                    item["page_url"],
                    item["title"],
                    item["publish_date"],
                    item["content"],
                    (item["content"] or "")[:500],
                    item["column"],
                )
            )
            if c.rowcount > 0:
                new_count += 1
            else:
                skip_count += 1
        except Exception as e:
            log(f"  DB error: {e}")
            skip_count += 1

    conn.commit()
    conn.close()
    return new_count, skip_count


def crawl_all(months_back=36):
    cutoff = (datetime.now() - timedelta(days=months_back * 30)).strftime("%Y-%m-%d")
    log(f"Full crawl (cutoff: {cutoff})")
    items = crawl_pages(0, 0, cutoff)
    enriched = process_details(items)
    new, skip = store_items(enriched)
    log(f"Result: {new} new, {skip} skipped")


def crawl_incremental():
    cutoff = (datetime.now() - timedelta(days=7)).strftime("%Y-%m-%d")
    log(f"Incremental (last 7 days)")
    items = crawl_pages(0, 0, cutoff)
    enriched = process_details(items)
    new, skip = store_items(enriched)
    log(f"Result: {new} new, {skip} skipped")


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        crawl_all()
    elif mode == "list":
        data = fetch_json(f"{APP_BASE}/gkmlpt/api/all/{CATEGORY_ID}?page=0&sid={SID}")
        arts = data.get("articles", []) if data else []
        log(f"Total: {len(arts)} articles")
        for a in arts[:5]:
            dt = datetime.fromtimestamp(a.get("date", 0)).strftime("%Y-%m-%d")
            print(f"  {dt} | {a.get('title','')[:60]}")
