#!/usr/bin/env python3
"""
围场满族蒙古族自治县人民政府 - 公告公示
URL: https://www.weichang.gov.cn/col/col10285/index.html?number=WC0004A00006
CMS: Hanweb (dataproxy.jsp AJAX分页)
List API: /module/web/jpage/dataproxy.jsp?page=N&appid=1&webid=30&columnid=10285&unitid=77347
Pagination: 74 pages, ~100条/页, 7392 total
Detail: div.cont(p正文) + meta ArticleTitle(标题)
"""
import sys, os, re, json, time
from datetime import datetime, timedelta
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

LIST_API = 'https://www.weichang.gov.cn/module/web/jpage/dataproxy.jsp'
SITE = "围场满族蒙古族自治县人民政府"
COLUMN = "公告公示"
PROVINCE = "河北"
TOTAL_PAGES = 74
DEFAULT_FULL_PAGES = 3

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)
BASE_URL = 'https://www.weichang.gov.cn'


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch_list(page):
    """通过 dataproxy.jsp API 获取列表（正则解析XML）"""
    params = {
        'page': page,
        'appid': 1,
        'webid': 30,
        'path': '/',
        'columnid': 10285,
        'unitid': 77347,
    }
    try:
        r = session.get(LIST_API, params=params, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None
        items = []
        # 用正则提取每个 record 里的 CDATA
        records = re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", r.text, re.DOTALL)
        for cdata in records:
            a_match = re.search(r"<a\s+href='([^']+)'\s*title='([^']+)'", cdata)
            span_match = re.search(r'<span[^>]*>([^<]+)</span>', cdata)
            if a_match and span_match:
                href = a_match.group(1).strip()
                title = a_match.group(2).strip()
                date_str = span_match.group(1).strip()
                full_url = urljoin(BASE_URL, href)
                items.append({'url': full_url, 'title': title, 'date': date_str})
        return items
    except Exception as e:
        log(f"  [ERROR] page={page}: {e}")
        return None


def extract_content(detail_url, list_title):
    try:
        r = session.get(detail_url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None, [], list_title
    except Exception as e:
        return None, [], list_title

    soup = BeautifulSoup(r.text, 'html.parser')

    # 标题: meta ArticleTitle > title tag
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        full_title = meta_title['content'].strip()
    else:
        title_tag = soup.find('title')
        full_title = title_tag.get_text(strip=True) if title_tag else list_title
        # 移除站点前缀 "围场满族蒙古族自治县人民政府 公告公示 "
        full_title = re.sub(r'^.*?县人民政府\s*\S*\s*', '', full_title)

    # 附件
    attachments = []
    for a in soup.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', re.I)):
        href = a.get('href', '').strip()
        name = a.get_text(strip=True) or '附件'
        if href:
            full_url = urljoin(detail_url, href)
            attachments.append({'name': name, 'url': full_url})

    # 正文: div.cont
    cont = soup.select_one('div.cont')
    if cont:
        # 剥离内联标签（span/b/strong/font 含文字样式）
        for tag in cont.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
            tag.unwrap()

        # 构建内容：表格 HTML 原样保留，段落按块提取
        parts = []
        for child in cont.children:
            if child.name == 'table':
                # 表格 HTML 原样保留（不剥离内联标签）
                html = str(child)
                if html.strip():
                    parts.append(html)
            elif child.name == 'p':
                text = child.get_text(separator='', strip=True)
                if text:
                    parts.append(text)
            elif child.name == 'h2':
                text = child.get_text(separator='', strip=True)
                if text:
                    parts.append(f'**{text}**')
            elif child.name == 'meta':
                continue
            elif child.name is None:
                # NavigableString (纯文本节点)，忽略
                continue
            else:
                # 其他块级元素（div等）-> 检查内部是否有表格
                tables_inner = child.find_all('table', recursive=False) if hasattr(child, 'find_all') else []
                if tables_inner:
                    # 如果内部有表格，保留表格 HTML，其余文本提取
                    for inner in child.children:
                        if inner.name == 'table':
                            html = str(inner)
                            if html.strip():
                                parts.append(html)
                        elif inner.name == 'p':
                            text = inner.get_text(separator='', strip=True)
                            if text:
                                parts.append(text)
                        elif inner.name is None:
                            continue
                        else:
                            text = inner.get_text(separator='', strip=True)
                            if text:
                                parts.append(text)
                else:
                    text = child.get_text(separator='', strip=True)
                    if text:
                        parts.append(text)
        content = '\n\n'.join(parts)

        # 空内容回退
        if len(content.strip()) < 20:
            content = ''
    else:
        content = ''

    # PDF空内容回退
    if not content.strip() and attachments:
        content = f'<p><a href="{detail_url}">{full_title}</a></p>'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'
    elif content.strip() and attachments:
        content += '\n\n**附件：**'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'

    return content, attachments, full_title


def crawl_pages(start_page, end_page, cutoff, label):
    all_items = []
    for p in range(start_page, end_page + 1):
        items = fetch_list(p)
        if not items:
            break
        log(f"  第{p}页: {len(items)}条")
        all_items.extend(items)
        time.sleep(0.5)

    log(f"\n[INFO] 共 {len(all_items)} 条列表项，开始爬详情...")
    total = len(all_items)
    for idx, item in enumerate(all_items):
        try:
            item_date = datetime.strptime(item['date'], '%Y-%m-%d')
            if item_date < cutoff:
                log(f"  [{idx+1}/{total}] ⏭️ 超出时间范围: {item['title'][:40]}")
                continue
        except ValueError:
            pass

        content, attachments, full_title = extract_content(item['url'], item['title'])
        if content is None:
            continue
        item['title'] = full_title
        item['content'] = content
        item['attachments'] = attachments
        output_item(item, idx)
        if idx % 20 == 0:
            log(f"  [PROGRESS] {idx}/{total}")
        time.sleep(0.3)
    log(f"\n[DONE] 共处理 {total} 条")


def crawl_all(months_back=36):
    cutoff = datetime.now() - timedelta(days=months_back * 30)
    end = min(TOTAL_PAGES, DEFAULT_FULL_PAGES)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 前{end}页 (新站策略, {months_back}个月)")
    log(f"{'='*50}")
    crawl_pages(1, end, cutoff, "full")


def crawl_incremental():
    cutoff = datetime.now() - timedelta(hours=48)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 增量（第1页）")
    log(f"{'='*50}")
    items = fetch_list(1)
    if not items:
        return
    log(f"第1页: {len(items)}条")
    for idx, item in enumerate(items):
        try:
            item_date = datetime.strptime(item['date'], '%Y-%m-%d')
            if item_date < cutoff:
                continue
        except ValueError:
            pass
        content, attachments, full_title = extract_content(item['url'], item['title'])
        if content is None:
            continue
        item['title'] = full_title
        item['content'] = content
        item['attachments'] = attachments
        output_item(item, idx)
        time.sleep(0.3)
    log("[DONE] 增量完成")


def output_item(item, idx):
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "attachments": item.get("attachments", []),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    print(json.dumps(record, ensure_ascii=False))


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        crawl_all()
    else:
        crawl_all()
