#!/usr/bin/env python3
"""Crawler for Jiangmen Guanghaiwan EZ - Bumen Wenjian
CapCloud CMS, static .shtml pages
Site: cnts.gov.cn / zfgzbm/tsghwgyyqglwyh/zwgk/zfxxgkml/bmwj/
"""

import requests
import re
import json
import sys
import os
import tempfile
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "http://www.cnts.gov.cn/zfgzbm/tsghwgyyqglwyh/zwgk/zfxxgkml/bmwj"
LIST_URL = BASE_URL + "/"
DETAIL_TEMPLATE = BASE_URL + "/content/post_{id}.html"
SITE_NAME = "台山市-江门市广海湾经济开发区管委会-部门文件"
GROUP = "广东组"
CATEGORY = "部门文件"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
TIMEOUT = 15
MAX_ITEMS = 200
YEAR_CUTOFF = 3

session = requests.Session()
session.headers.update(HEADERS)


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for div in soup.find_all("div", class_="list_div"):
        title_div = div.find("div", class_="list_right_title")
        if not title_div:
            continue
        a = title_div.find("a")
        if not a or not a.get("href"):
            continue
        href = a["href"].strip()
        title = a.get_text(strip=True)
        if not title or not href:
            continue
        m = re.search(r'post_(\d+)', href)
        if m:
            items.append((int(m.group(1)), title, href))
    return items


def parse_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    title_tag = soup.find("ucaptitle")
    title = title_tag.get_text(strip=True) if title_tag else ""
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta:
            title = meta.get("content", "")
    date_tag = soup.find("publishtime")
    pub_date = ""
    if date_tag:
        pub_date = date_tag.get_text(strip=True)
    if not pub_date:
        meta = soup.find("meta", attrs={"name": "PubDate"})
        if meta:
            pub_date = meta.get("content", "")
    content_tag = soup.find("ucapcontent")
    content_html = ""
    if content_tag:
        content_html = str(content_tag)

    attachments = []
    zoom_div = soup.find("div", id="zoom")
    if zoom_div:
        for a in zoom_div.find_all("a", href=True):
            ahref = a["href"].strip().lower()
            if any(ext in ahref for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", "downfile"]):
                attachments.append({
                    "title": a.get_text(strip=True) or "附件",
                    "url": a["href"]
                })
    return {
        "title": title,
        "date": pub_date[:10] if pub_date else "",
        "content": content_html,
        "attachments": attachments,
        "source_url": url
    }


def is_incremental():
    return len(sys.argv) > 1 and sys.argv[1] == "incremental"


def main():
    incremental = is_incremental()
    print(f"[{SITE_NAME}] Starting crawl (incremental={incremental})")
    print(f"  Fetching list: {LIST_URL}")
    try:
        resp = session.get(LIST_URL, timeout=TIMEOUT)
        resp.encoding = "utf-8"
    except Exception as e:
        print(f"  ERROR fetching list: {e}")
        sys.exit(1)
    items = parse_list(resp.text)
    print(f"  Found {len(items)} items on list page")
    items = items[:MAX_ITEMS]
    if not items:
        print("  No items found, exiting")
        return
    cutoff_date = None
    if incremental:
        from datetime import timedelta
        cutoff_date = datetime.now() - timedelta(days=YEAR_CUTOFF * 365)
        print(f"  Incremental mode: cutoff date = {cutoff_date.date()}")
    results = []
    for idx, (post_id, title, url) in enumerate(items):
        detail_url = url if url.startswith("http") else DETAIL_TEMPLATE.format(id=post_id)
        print(f"  [{idx+1}/{len(items)}] {title[:40]}...")
        try:
            resp = session.get(detail_url, timeout=TIMEOUT)
            resp.encoding = "utf-8"
        except Exception as e:
            print(f"    ERROR: {e}")
            continue
        detail = parse_detail(resp.text, detail_url)
        if not detail or not detail["content"]:
            detail = detail or {}
            detail["content"] = f'<p><a href="{detail_url}">{title}</a></p>'
            detail["title"] = title
        if incremental and cutoff_date and detail.get("date"):
            try:
                item_date = datetime.strptime(detail["date"], "%Y-%m-%d")
                if item_date < cutoff_date:
                    print(f"    Skipped (too old: {detail['date']})")
                    continue
            except ValueError:
                pass
        results.append(detail)
    print(f"\n  Total valid items: {len(results)}")
    if not results:
        print("  No items to import, exiting")
        return

    # Write JSONL
    tmp = tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=False, dir="/tmp")
    label = f"{SITE_NAME} ({len(results)}条)"
    for r in results:
        line = {
            "title": r["title"],
            "page_url": r["source_url"],
            "content": r["content"],
            "publish_date": r["date"],
            "site_name": SITE_NAME,
            "source_url": r["source_url"],
            "attachments": r["attachments"]
        }
        tmp.write(json.dumps(line, ensure_ascii=False) + "\n")
    tmp.close()
    print(f"  JSONL written: {tmp.name}")

    # Import
    ret = os.system(f"python3 /root/gov_crawler/import_jsonl.py {tmp.name} '{label}'")
    if ret == 0:
        print(f"  OK Import complete")
    else:
        print(f"  FAIL Import failed (exit={ret})")
        sys.exit(1)


if __name__ == "__main__":
    main()
