#!/usr/bin/env python3
"""大柴旦行政委员会 - 公示公告 爬虫
URL: http://www.dachaidan.gov.cn/xwdt/gsgg.htm
CMS: Visual SiteBuilder 9 (VSB)
分页: gsgg.htm (第1页) + gsgg/20.htm ~ gsgg/1.htm (第2~21页, 共518条)
详情: /info/1065/{id}.htm → div#vsb_content div.v_news_content
"""

import sys
import os
import re
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

BASE_URL = "http://www.dachaidan.gov.cn"
LIST_URL = "http://www.dachaidan.gov.cn/xwdt/gsgg.htm"
COLUMN = "公示公告"
SITE = "大柴旦行政委员会"
PROVINCE = "青海"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

session = requests.Session()
session.headers.update(HEADERS)


def log(msg):
    sys.stderr.write(msg + "\n")
    sys.stderr.flush()


def extract_list(page_url):
    """Extract (title, url, date) from a list page."""
    try:
        resp = session.get(page_url, timeout=15)
        resp.encoding = "utf-8"
    except Exception as e:
        log(f"  [ERROR] 请求列表页失败: {page_url} - {e}")
        return []

    soup = BeautifulSoup(resp.text, "html.parser")
    items = []
    for tr in soup.select("tr"):
        link = tr.select_one("a.c3158")
        if not link or not link.get("href"):
            continue
        href = link.get("href", "")
        if "../info/" not in href:
            continue
        title = link.get("title", "").strip() or link.get_text(strip=True)
        if not title:
            continue
        url = urljoin(BASE_URL, href)
        date_span = tr.select_one("span.timestyle3158")
        date_str = date_span.get_text(strip=True) if date_span else ""
        # Normalize date YYYY/MM/DD → YYYY-MM-DD
        date_str = date_str.replace("/", "-").strip()
        items.append({"title": title, "url": url, "date": date_str})
    return items


def extract_detail(detail_url):
    """Extract content from detail page."""
    try:
        resp = session.get(detail_url, timeout=15)
        resp.encoding = "utf-8"
    except Exception as e:
        log(f"  [ERROR] 请求详情页失败: {detail_url} - {e}")
        return "", "", ""

    soup = BeautifulSoup(resp.text, "html.parser")

    # Title
    title_el = soup.select_one("td.titlestyle3191")
    title = title_el.get_text(strip=True) if title_el else ""

    # Date
    date_el = soup.select_one("span.timestyle3191")
    date_str = date_el.get_text(strip=True) if date_el else ""

    # Content
    content_div = soup.select_one("#vsb_content div.v_news_content")
    if not content_div:
        content_div = soup.select_one("#vsb_content")
    if not content_div:
        content_div = soup.select_one("div.c3191_content")

    content = ""
    if content_div:
        # Preserve table HTML before extraction
        table_htmls = []
        for table in content_div.find_all("table"):
            table_htmls.append(str(table))
            table.decompose()

        # Insert paragraph markers at p boundaries
        for p_tag in content_div.find_all("p"):
            p_tag.append("\u00b6P\u00b6")

        # Strip inner inline tags (prevents date/number splitting)
        for tag in content_div.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
            tag.unwrap()

        # Use empty separator so adjacent text merges, then restore paragraph breaks
        content = content_div.get_text(separator="", strip=True)
        content = content.replace("\u00b6P\u00b6", "\n\n")
        content = re.sub(r"\n{3,}", "\n\n", content)

        # Append table HTML at end
        for th in table_htmls:
            content += "\n\n" + th

    # Check for PDF-only / empty content# Check for PDF-only / empty content
    pdf_links = content_div.find_all("a", href=re.compile(r"\.pdf", re.I)) if content_div else []
    attachments = []
    for a_el in content_div.find_all("a", href=True) if content_div else []:
        href = a_el.get("href", "")
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", href, re.I):
            attachments.append(urljoin(BASE_URL, href))

    if not content.strip() or len(content.strip()) < 20:
        if pdf_links or attachments:
            content = f'<p><a href="{detail_url}">{title}</a></p>'
        elif title:
            content = f'<p><a href="{detail_url}">{title}</a></p>'

    return title, date_str, content


def crawl_all(months_back=36):
    """Full crawl: scrape all 21 pages then detail."""
    cutoff = datetime.now() - timedelta(days=months_back * 30) if months_back else None

    # Collect all list page URLs
    # Page 1 (newest): gsgg.htm
    # Page 2..21: gsgg/20.htm .. gsgg/1.htm
    list_urls = [LIST_URL]
    for i in range(20, 0, -1):
        list_urls.append(f"http://www.dachaidan.gov.cn/xwdt/gsgg/{i}.htm")

    all_items = []
    seen_urls = set()
    for page_url in list_urls:
        items = extract_list(page_url)
        if not items:
            print(f"  [SKIP] 空页: {page_url}", flush=True)
            continue
        new_count = 0
        for item in items:
            if item["url"] not in seen_urls:
                if cutoff and item["date"]:
                    try:
                        item_date = datetime.strptime(item["date"][:10], "%Y-%m-%d")
                        if item_date < cutoff:
                            continue
                    except ValueError:
                        pass
                seen_urls.add(item["url"])
                all_items.append(item)
                new_count += 1
        log(f"  [LIST] {page_url} → +{new_count} 条")

    log(f"\n共 {len(all_items)} 条待爬详情")

    for idx, item in enumerate(all_items, 1):
        title, date_str, content = extract_detail(item["url"])
        # Use extracted date if list date was empty
        if not item["date"] and date_str:
            item["date"] = date_str.replace("/", "-").strip()
        if title:
            item["title"] = title
        item["content"] = content

        # Output for pipeline ingestion
        output_item(item, idx)
        if idx % 10 == 0:
            log(f"  [PROGRESS] {idx}/{len(all_items)}")

    log(f"\n[DONE] 共爬取 {len(all_items)} 条")


def crawl_incremental():
    """Incremental crawl: only the first page (newest 25 items)."""
    items = extract_list(LIST_URL)
    log(f"[LIST] {LIST_URL} → {len(items)} 条")

    for idx, item in enumerate(items, 1):
        title, date_str, content = extract_detail(item["url"])
        if not item["date"] and date_str:
            item["date"] = date_str.replace("/", "-").strip()
        if title:
            item["title"] = title
        item["content"] = content
        output_item(item, idx)

    log(f"\n[DONE] 增量爬取 {len(items)} 条")


def output_item(item, idx):
    """Output in pipeline-compatible JSONL format (import_jsonl.py compatible) to stdout."""
    import json
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    print(json.dumps(record, ensure_ascii=False))


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        months = int(sys.argv[2]) if len(sys.argv) > 2 else 36
        crawl_all(months)
    elif mode == "list":
        items = extract_list(LIST_URL)
        log(f"共 {len(items)} 条")
        for it in items[:5]:
            log(f"  {it['date']} | {it['title'][:40]} | {it['url']}")
    else:
        log(f"Usage: {sys.argv[0]} [incremental|full [months]]")
