#!/usr/bin/env python3
"""
Crawler for 察布查尔县人民政府 - 公示公告
CMS: Visual SiteBuilder 9 (VSB), static .shtml pages
Site: www.xjcbcr.gov.cn/xjcbcr/c114052/
"""
import requests, re, json, sys, os, tempfile
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.xjcbcr.gov.cn"
LIST_PATH = "/xjcbcr/c114052"
LIST_URL = BASE_URL + LIST_PATH + "/list.shtml"
SITE_NAME = "察布查尔县人民政府-公示公告"
INCREMENTAL_DAYS = 7
MAX_PAGES = 5
PER_PAGE = 7

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20

session = requests.Session()
session.headers.update(HEADERS)


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def fetch(url):
    resp = session.get(url, timeout=TIMEOUT)
    resp.encoding = "utf-8"
    return resp.text


def parse_list(html):
    """Parse list page, return list of (title, url, date)."""
    soup = BeautifulSoup(html, "html.parser")
    items = []

    # Content is in div.p10 > table with rows
    p10 = soup.find("div", class_="p10")
    if not p10:
        return items

    table = p10.find("table")
    if not table:
        return items

    for tr in table.find_all("tr"):
        cells = tr.find_all("td")
        if len(cells) < 2:
            continue

        # Find article link: <a class="c1457"> (not the column nav link)
        a = tr.find("a", class_="c1457", href=True)
        if not a:
            a = tr.find("a", href=True)
            if not a:
                continue
            # Skip if it's a nav link (column selector)
            href = a["href"].strip()
            if "c114052" in href or "common_list" in href:
                continue
        else:
            href = a["href"].strip()

        # Title from title attribute (full title, no ellipsis!)
        title = a.get("title") or a.get_text(strip=True)
        if not title:
            continue

        # Date from the second-to-last or third cell
        date = cells[2].get_text(strip=True) if len(cells) > 2 else cells[1].get_text(strip=True)

        # Resolve relative URL
        if href.startswith("/"):
            href = BASE_URL + href
        elif href.startswith("./"):
            href = BASE_URL + LIST_PATH + "/" + href[2:]
        elif not href.startswith("http"):
            href = BASE_URL + LIST_PATH + "/" + href

        if title and href:
            items.append((title.strip(), href, date[:10]))
    return items


def parse_detail(html, url):
    """Parse detail page content."""
    soup = BeautifulSoup(html, "html.parser")

    # Title from h1
    h1 = soup.find("h1")
    title = h1.get_text(strip=True) if h1 else ""

    # Date from the meta div inside div.p15
    pub_date = ""
    p15 = soup.find("div", class_="p15")
    if p15:
        meta_div = p15.find("div", style=re.compile(r"font-size.*14"))
        if meta_div:
            txt = meta_div.get_text(strip=True)
            m = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
            if m:
                pub_date = m.group(1)

    # Content from vsb_content > v_news_content
    content_html = ""
    vsb = soup.find("div", id="vsb_content") or soup.find("div", class_="vsb_content")
    if vsb:
        vnc = vsb.find("div", class_="v_news_content")
        if vnc:
            content_html = str(vnc)

    # Attachments
    attachments = []
    for a in soup.find_all("a", href=True):
        ahref = a["href"].strip().lower()
        if ahref.endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
            full_url = a["href"].strip()
            if not full_url.startswith("http"):
                full_url = BASE_URL + full_url
            attachments.append({
                "title": a.get_text(strip=True) or "附件",
                "url": full_url
            })

    return {
        "title": title,
        "date": pub_date,
        "content": content_html,
        "attachments": attachments,
        "source_url": url,
    }


def render_content(content_html, title, url):
    """Convert VSB content HTML to readable markdown with tables."""
    if not content_html or len(content_html.strip()) < 50:
        return "[{}]({})".format(title, url)

    soup = BeautifulSoup(content_html, "html.parser")
    parts = []

    for el in soup.find_all(["p", "table", "h1", "h2", "h3", "h4", "img"], recursive=True):
        if el.name == "p":
            text = el.get_text(" ", strip=True)
            imgs = el.find_all("img")
            if imgs:
                for img in imgs:
                    src = img.get("src", "")
                    if src:
                        src = urljoin(url, src) if not src.startswith("http") else src
                        parts.append("![图片]({})".format(src))
            if text:
                parts.append(text)
        elif el.name == 'table':
            tbl_html = html_table_to_html(el, url)
            if tbl_html:
                parts.append(tbl_html)
        elif el.name in ("h1", "h2", "h3", "h4"):
            text = el.get_text(" ", strip=True)
            if text:
                parts.append("\n{}".format(text))
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                src = urljoin(url, src) if not src.startswith("http") else src
                parts.append("![图片]({})".format(src))

    result = "\n\n".join(p for p in parts if p).strip()
    result = re.sub(r"\n{3,}", "\n\n", result)

    if len(result.strip()) < 20:
        return "[{}]({})".format(title, url)
    return result


def is_incremental():
    return len(sys.argv) > 1 and sys.argv[1] == "incremental"


def main():
    incremental = is_incremental()
    print("[{}] Starting crawl (incremental={}, max_pages={})".format(SITE_NAME, incremental, MAX_PAGES))

    # Fetch list pages
    all_items = []
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            page_url = LIST_URL
        else:
            page_url = "{}/list_{}.shtml".format(LIST_URL.rstrip("/").replace("list.shtml", ""), page)
        print("  Fetching page {}: {}".format(page, page_url))
        try:
            html = fetch(page_url)
        except Exception as e:
            print("    ERROR: {}".format(e))
            continue
        items = parse_list(html)
        print("    Found {} items".format(len(items)))
        if not items:
            print("    No more items, stopping")
            break
        all_items.extend(items)

    print("\n  Total items from list: {}".format(len(all_items)))
    if not all_items:
        print("  No items found, exiting")
        return

    # Incremental filter
    cutoff_date = None
    if incremental:
        cutoff_date = datetime.now() - timedelta(days=INCREMENTAL_DAYS)
        print("  Incremental mode: cutoff = {}".format(cutoff_date.date()))
        filtered = [(t, u, d) for t, u, d in all_items if d and date_filter(d, cutoff_date)]
        all_items = filtered
        print("  After incremental filter: {} items".format(len(all_items)))

    # Fetch details
    results = []
    for idx, (title, url, date_from_list) in enumerate(all_items):
        print("  [{}/{}] Fetching: {}...".format(idx + 1, len(all_items), title[:40]))
        try:
            html = fetch(url)
        except Exception as e:
            print("    ERROR: {}".format(e))
            results.append({
                "title": title, "date": date_from_list,
                "content": "[{}]({})".format(title, url),
                "attachments": [], "source_url": url,
            })
            continue

        detail = parse_detail(html, url)
        if not detail["title"]:
            detail["title"] = title
        if not detail["date"]:
            detail["date"] = date_from_list

        detail["content"] = render_content(detail["content"], detail["title"], url)
        results.append(detail)

    print("\n  Total valid items: {}".format(len(results)))
    if not results:
        print("  No items to import, exiting")
        return

    # Write JSONL
    tmp = tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=False, dir="/tmp")
    label = "{} ({}条)".format(SITE_NAME, len(results))
    for r in results:
        line = {
            "title": r["title"],
            "page_url": r["source_url"],
            "content": r["content"],
            "publish_date": r["date"],
            "site_name": SITE_NAME,
            "source_url": r["source_url"],
            "attachments": r.get("attachments", []),
        }
        tmp.write(json.dumps(line, ensure_ascii=False) + "\n")
    tmp.close()
    print("  JSONL written: {}".format(tmp.name))

    # Import
    ret = os.system("python3 /root/gov_crawler/import_jsonl.py '{}' '{}'".format(tmp.name, label))
    if ret == 0:
        print("  OK Import complete")
    else:
        print("  FAIL Import failed (exit={})".format(ret))
        sys.exit(1)

    # import_jsonl.py already removes the temp file, so only try if it still exists
    if os.path.exists(tmp.name):
        try:
            os.unlink(tmp.name)
        except OSError:
            pass


def date_filter(date_str, cutoff):
    try:
        d = datetime.strptime(date_str, "%Y-%m-%d")
        return d >= cutoff
    except ValueError:
        return True


if __name__ == "__main__":
    main()
