#!/usr/bin/env python3
"""
Crawler: 芜湖市弋江区生态环境分局 - 行政审批 (Epoint 基层政务公开, catId=1141240)
Fix: content is in yjq.gov.cn page (j-fontContent gkwz_contnet), preserving tables as HTML.
"""

import sys, os, json, re, time, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE = "弋江区生态环境分局行政审批"
COLUMN = "行政审批"
BASE_URL = "https://www.yjq.gov.cn"
STHJJ_URL = "https://sthjj.wuhu.gov.cn"
API_URL = f"{BASE_URL}/wuhu/site/label/8888"
API_DATA = {
    "labelName": "grassrootsContentPageList",
    "siteId": "6787891",
    "organId": "6603431",
    "pageSize": "15",
    "pageIndex": "1",
    "isDate": "true",
    "dateFormat": "yyyy-MM-dd",
    "length": "50",
    "type": "4",
    "action": "list",
    "isJson": "true",
    "catId": "1141240",
}
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
}


def log(msg):
    print(f"[{SITE}] {msg}", flush=True)


def fetch_list_page(page_index):
    data = {**API_DATA, "pageIndex": str(page_index)}
    try:
        r = requests.post(API_URL, data=data, headers=HEADERS, timeout=30, verify=False)
        return r.json()
    except Exception as e:
        log(f"API error: {e}")
        return None


def fetch_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        return r.text, detail_url
    except Exception as e:
        log(f"Fetch error for {detail_url}: {e}")
        return None, detail_url


def get_content_by_class(html):
    """Extract content from j-fontContent div. Works for both gkwz_contnet and newscontnet minh500."""
    soup = BeautifulSoup(html, "html.parser")
    # Find any div with j-fontContent in class
    content_div = soup.find("div", class_=lambda c: c and "j-fontContent" in c.split())
    if content_div:
        return _clean_content(str(content_div))
    return ""


def _clean_content(content_html):
    """Convert content HTML to text. Preserve tables as HTML in original position. separator='' for paragraphs."""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")

    # Convert attachment links to markdown
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        full_url = urljoin(BASE_URL, href)
        a_tag.replace_with(f"[{text}]({full_url})")

    # Unwrap inline formatting tags
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()

    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            t = el.get_text(separator='', strip=True)
            if t:
                parts.append(t)
    return '\n\n'.join(parts) if parts else content_html.strip()


def parse_detail(html):
    """Extract title, date, content from detail page."""
    result = {"title": "", "publish_date": "", "content": ""}
    m = re.search(r'<title>(.*?)</title>', html)
    if m:
        title = m.group(1).strip()
        title = re.sub(r'[_\-—]\s*芜湖市生态环境局\s*$', '', title).strip()
        title = re.sub(r'[_\-—]\s*芜湖市政务公开平台\s*$', '', title).strip()
        result["title"] = title
    m = re.search(r'PubDate.*?content="([^"]+)"', html)
    if m:
        result["publish_date"] = m.group(1).strip()[:10]
    result["content"] = get_content_by_class(html)
    return result


def crawl_all(months_back=36):
    cutoff = (datetime.now() - timedelta(days=months_back * 30)).strftime("%Y-%m-%d")
    log(f"Full crawl (cutoff: {cutoff})")
    data = fetch_list_page(1)
    if not data:
        return
    total = data.get("total", 0)
    page_size = int(API_DATA["pageSize"])
    total_pages = (total + page_size - 1) // page_size
    log(f"Total: {total}, pages: {total_pages}")

    all_items = []
    for pg in range(1, min(total_pages + 1, 501)):
        data = fetch_list_page(pg)
        if not data:
            break
        items = data.get("data", [])
        for item in items:
            pub_date = (item.get("publishDate") or "")[:10]
            if pub_date < cutoff:
                continue
            detail_url = item.get("urlDomain", "") or item.get("link", "") or f"{BASE_URL}/openness/grassroots/{item['contentId']}.html"
            all_items.append({
                "content_id": item["contentId"],
                "title": item.get("title", ""),
                "date": pub_date,
                "url": detail_url,
            })
        log(f"  Page {pg}: {len(items)} items")
        time.sleep(0.5)

    log(f"After filter: {len(all_items)} items")

    enriched = []
    for idx, item in enumerate(all_items, 1):
        html, _ = fetch_detail(item["url"])
        if html:
            detail = parse_detail(html)
            if detail["title"]:
                item["title"] = detail["title"]
            if detail["publish_date"]:
                item["date"] = detail["publish_date"]
            item["content"] = detail["content"]
        else:
            item["content"] = ""

        enriched.append({
            "title": item["title"],
            "page_url": item["url"],
            "publish_date": item["date"],
            "content": item.get("content", ""),
            "site_name": f"{SITE}-{COLUMN}",
            "source_url": item["url"],
        })
        if idx % 10 == 0:
            log(f"  Detail: {idx}/{len(all_items)}")
        time.sleep(0.5)

    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], COLUMN))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


def crawl_first_pages(pages=5):
    """Full crawl: fetch first N pages, no date cutoff."""
    log(f"Full first {pages} pages (no date cutoff)")
    data = fetch_list_page(1)
    if not data:
        return
    total = data.get("total", 0)
    page_size = int(API_DATA["pageSize"])
    total_pages = min((total + page_size - 1) // page_size, pages)
    log(f"Total: {total}, crawling first {total_pages} pages")

    all_items = []
    for pg in range(1, total_pages + 1):
        data = fetch_list_page(pg)
        if not data:
            break
        items = data.get("data", [])
        for item in items:
            detail_url = item.get("urlDomain", "") or item.get("link", "") or f"{BASE_URL}/openness/grassroots/{item['contentId']}.html"
            pub_date = (item.get("publishDate") or "")[:10]
            all_items.append({
                "content_id": item["contentId"],
                "title": item.get("title", ""),
                "date": pub_date,
                "url": detail_url,
            })
        log(f"  Page {pg}: {len(items)} items")
        time.sleep(0.5)

    log(f"Total items: {len(all_items)}")

    enriched = []
    for idx, item in enumerate(all_items, 1):
        html, _ = fetch_detail(item["url"])
        if html:
            detail = parse_detail(html)
            if detail["title"]:
                item["title"] = detail["title"]
            if detail["publish_date"]:
                item["date"] = detail["publish_date"]
            item["content"] = detail["content"]
        else:
            item["content"] = ""

        enriched.append({
            "title": item["title"],
            "page_url": item["url"],
            "publish_date": item["date"],
            "content": item.get("content", ""),
            "site_name": f"{SITE}-{COLUMN}",
            "source_url": item["url"],
        })
        if idx % 10 == 0:
            log(f"  Detail: {idx}/{len(all_items)}")
        time.sleep(0.5)

    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], COLUMN))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


def crawl_incremental():
    cutoff = (datetime.now() - timedelta(days=7)).strftime("%Y-%m-%d")
    log(f"Incremental (since {cutoff})")
    data = fetch_list_page(1)
    if not data:
        return
    items = data.get("data", [])
    filtered = [it for it in items if (it.get("publishDate") or "")[:10] >= cutoff]

    enriched = []
    for item in filtered:
        detail_url = item.get("urlDomain", "") or item.get("link", "") or f"{BASE_URL}/openness/grassroots/{item['contentId']}.html"
        html, _ = fetch_detail(detail_url)
        detail = parse_detail(html) if html else {"title": "", "publish_date": "", "content": ""}
        enriched.append({
            "title": detail["title"] or item.get("title", ""),
            "page_url": detail_url,
            "publish_date": (item.get("publishDate") or "")[:10],
            "content": detail["content"],
            "site_name": f"{SITE}-{COLUMN}",
            "source_url": detail_url,
        })
        time.sleep(0.5)

    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], COLUMN))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode in ("full", "all"):
        crawl_first_pages(pages=999)
    elif mode == "list":
        data = fetch_list_page(1)
        total = data.get("total", 0) if data else 0
        log(f"Total: {total}")
