#!/usr/bin/env python3
"""柏乡县人民政府 - 公告公示 爬虫
URL: https://www.baixiangxian.gov.cn/channel/list/77.html
CMS: TopwayCMS
分页: list/77.html (第1页) + list/77_{N}.html (2-10) + list_xxgk.jsp (11~145)
详情: /single/77/{id}.html -> div.detail-2
"""

import sys
import os
import re
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

BASE_URL = "https://www.baixiangxian.gov.cn"
LIST_URL = "https://www.baixiangxian.gov.cn/channel/list/77.html"
COLUMN = "公告公示"
SITE = "柏乡县人民政府"
PROVINCE = "河北"
CLASS_ID = 77

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

session = requests.Session()
session.headers.update(HEADERS)

MARKER = "<<P>>"


def log(msg):
    sys.stderr.write(msg + "\n")
    sys.stderr.flush()


def extract_list(page_url):
    """Extract (title, url, date) from a list page."""
    try:
        resp = session.get(page_url, timeout=15)
        resp.encoding = "utf-8"
    except Exception as e:
        log(f"  [ERROR] 请求列表页失败: {page_url} - {e}")
        return []

    soup = BeautifulSoup(resp.text, "html.parser")
    items = []
    for li in soup.select("ul.info-list-xxgk > li"):
        link = li.select_one("a[href^='/single/']")
        if not link:
            continue
        href = link.get("href", "").strip()
        title = link.get("title", "") or link.get_text(strip=True)
        if not title:
            continue
        url = urljoin(BASE_URL, href)
        time_el = li.select_one("span.time")
        date_str = time_el.get_text(strip=True) if time_el else ""
        items.append({"title": title, "url": url, "date": date_str})
    return items


def get_list_urls():
    """Generate all list page URLs for pages 1-145."""
    urls = [LIST_URL]
    for n in range(2, 11):
        urls.append(f"{BASE_URL}/channel/list/77_{n}.html")
    for n in range(11, 146):
        urls.append(f"{BASE_URL}/xxgk/list/list_xxgk.jsp?classId={CLASS_ID}&&pn={n}")
    return urls


def extract_detail(detail_url):
    """Extract content from detail page with paragraph and table fixes."""
    try:
        resp = session.get(detail_url, timeout=15)
        resp.encoding = "utf-8"
    except Exception as e:
        log(f"  [ERROR] 请求详情页失败: {detail_url} - {e}")
        return "", "", "", []

    soup = BeautifulSoup(resp.text, "html.parser")

    # Title from h1
    title_el = soup.select_one("h1.t-type2")
    title = title_el.get_text(strip=True) if title_el else ""

    # Date from meta PubDate first
    date_str = ""
    meta_date = soup.select_one("meta[name=PubDate]")
    if meta_date and meta_date.get("content"):
        date_str = meta_date["content"].strip()
    if not date_str:
        info_el = soup.select_one("p.infoi")
        if info_el:
            text = info_el.get_text()
            m = re.search(r'发布时间：(\d{4}-\d{2}-\d{2})', text)
            if m:
                date_str = m.group(1)

    # Attachments
    attachments = []
    content_div = soup.select_one("div#content.detail-2")
    if not content_div:
        content_div = soup.select_one("div.detail-2")
    if content_div:
        for a in content_div.find_all("a", href=True):
            href = a["href"].strip()
            if href.endswith(('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar')):
                text = a.get_text(strip=True)
                full_url = urljoin(BASE_URL, href) if href.startswith('/') else href
                attachments.append(f'<p><a href="{full_url}">{text}</a></p>')

    content = extract_content(content_div, title, detail_url)
    return title, date_str, content, attachments


def extract_content(content_div, title, detail_url):
    """Extract formatted content with proper paragraphs and table preservation."""
    if not content_div:
        if title:
            return f'<p><a href="{detail_url}">{title}</a></p>'
        return ""

    # 1) Save and remove tables
    tables_html = []
    for table in content_div.find_all("table"):
        tables_html.append(str(table))
        table.decompose()

    # 2) Insert paragraph markers for <p>, headings, and <br>
    for tag in content_div.find_all(['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6']):
        tag.insert(0, MARKER)
        tag.append(MARKER)
    for br in content_div.find_all('br'):
        br.replace_with(MARKER)

    # 3) Unwrap inline tags with empty separator
    for tag in content_div.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
        tag.unwrap()

    # 4) Extract text with empty separator (no span-induced splits)
    text = content_div.get_text(separator="", strip=True)

    # 5) Clean up markers
    text = re.sub(r'<<P>>(\s*<<P>>)+', '<<P>>', text)
    text = text.replace('<<P>>', '\n\n')
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = text.strip()

    # 6) Append tables at end
    for tbl_html in tables_html:
        text += f"\n\n{tbl_html}"

    # 7) Empty content fallback
    if not text.strip() or len(text.strip()) < 20:
        if title:
            text = f'<p><a href="{detail_url}">{title}</a></p>'

    return text


def crawl_all(months_back=36):
    """Full crawl: scrape all pages then detail."""
    cutoff = datetime.now() - timedelta(days=months_back * 30) if months_back else None

    list_urls = get_list_urls()
    all_items = []
    seen_urls = set()

    for page_url in list_urls:
        items = extract_list(page_url)
        if not items:
            continue
        new_count = 0
        for item in items:
            if item["url"] not in seen_urls:
                if cutoff and item["date"]:
                    try:
                        item_date = datetime.strptime(item["date"], "%Y-%m-%d")
                        if item_date < cutoff:
                            continue
                    except ValueError:
                        pass
                seen_urls.add(item["url"])
                all_items.append(item)
                new_count += 1
        log(f"  [LIST] {page_url[:80]} -> +{new_count} 条")
        if new_count == 0 and len(all_items) > 0 and cutoff:
            log(f"  [STOP] 遇到早于截止日期的页面")
            break

    log(f"\n共 {len(all_items)} 条待爬详情")

    for idx, item in enumerate(all_items, 1):
        title, date_str, content, attachments = extract_detail(item["url"])
        if not item["date"] and date_str:
            item["date"] = date_str
        if title:
            item["title"] = title
        item["content"] = content
        if attachments:
            item["attachments"] = attachments
        output_item(item, idx)
        if idx % 10 == 0:
            log(f"  [PROGRESS] {idx}/{len(all_items)}")

    log(f"\n[DONE] 共爬取 {len(all_items)} 条")


def crawl_incremental():
    """Incremental crawl: only the first page (newest items)."""
    items = extract_list(LIST_URL)
    log(f"[LIST] {LIST_URL} -> {len(items)} 条")

    for idx, item in enumerate(items, 1):
        title, date_str, content, attachments = extract_detail(item["url"])
        if not item["date"] and date_str:
            item["date"] = date_str
        if title:
            item["title"] = title
        item["content"] = content
        if attachments:
            item["attachments"] = attachments
        output_item(item, idx)

    log(f"\n[DONE] 增量爬取 {len(items)} 条")


def output_item(item, idx):
    """Output JSONL to stdout for pipeline ingestion."""
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    if item.get("attachments"):
        record["attachments"] = json.dumps(item["attachments"], ensure_ascii=False)
    print(json.dumps(record, ensure_ascii=False))


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        months = int(sys.argv[2]) if len(sys.argv) > 2 else 36
        crawl_all(months)
    elif mode == "list":
        items = extract_list(LIST_URL)
        log(f"共 {len(items)} 条")
        for it in items[:5]:
            log(f"  {it['date']} | {it['title'][:40]} | {it['url']}")
    else:
        log(f"Usage: {sys.argv[0]} [incremental|full [months]]")
