#!/usr/bin/env python3
"""
云浮市生态环境局 - 受理公告
URL: https://www.yunfu.gov.cn/sthjj/zdlyxxgkzl/jsxmhjyxpj/slgg/
CMS: TRS
分页: index.html / index_N.html (4页有数据, ~86条)
列表: div.nyrtct ul > li > span(日期) + a(标题+链接)
详情: div.TRS_Editor 正文 + 附件
"""
import sys, os, re, json, time
from datetime import datetime, timedelta
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.yunfu.gov.cn/sthjj/zdlyxxgkzl/jsxmhjyxpj/slgg/'
LIST_URL = BASE_URL + 'index.html'
SITE = "云浮市生态环境局"
COLUMN = "受理公告"
PROVINCE = "广东"
TOTAL_PAGES = 5  # index.html + index_2~5
DEFAULT_FULL_PAGES = 5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)

ATTACH_EXTS = ('.doc', '.docx', '.pdf', '.xls', '.xlsx', '.ppt', '.pptx',
               '.zip', '.rar', '.7z', '.tar', '.gz', '.txt',)


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def is_attachment(href):
    href = href.lower().strip()
    return any(href.endswith(ext) for ext in ATTACH_EXTS) or '/attachment/' in href


def fetch_list_html(page):
    """Fetch list page HTML"""
    if page == 1:
        url = LIST_URL
    else:
        url = LIST_URL.replace('index.html', f'index_{page}.html')
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            log(f"  [WARN] 第{page}页 status={r.status_code}")
            return None
        return r.text
    except Exception as e:
        log(f"  [ERROR] 第{page}页: {e}")
        return None


def parse_list(html):
    """Parse list items from HTML"""
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.select_one('div.nyrtct ul')
    if not ul:
        return []
    items = []
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        span = li.find('span')
        if not a or not span:
            continue
        href = a.get('href', '').strip()
        title = a.get('title', '').strip() or a.get_text(strip=True)
        date_str = span.get_text(strip=True)
        if not href or not title:
            continue
        full_url = urljoin(LIST_URL, href)
        items.append({'url': full_url, 'title': title, 'date': date_str})
    return items


def extract_content(detail_url, list_title):
    """提取正文、附件和完整标题"""
    try:
        r = session.get(detail_url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            log(f"  [WARN] {detail_url} status={r.status_code}")
            return None, [], list_title
    except Exception as e:
        log(f"  [ERROR] {detail_url}: {e}")
        return None, [], list_title

    soup = BeautifulSoup(r.text, 'html.parser')

    # 从详情页提取完整标题（列表页标题可能被截断）
    h2 = soup.find('h2')
    full_title = h2.get_text(strip=True) if h2 else list_title

    # 收集附件（在TRS_Editor外部或内部的doc链接）
    attachments = []
    # 收集附件：匹配所有带扩展名的附件链接和class=doc/nfw-cms-attachment的链接
    for a in soup.find_all('a'):
        href = a.get('href', '').strip()
        if not href:
            continue
        cls = ' '.join(a.get('class', [])) if a.get('class') else ''
        if is_attachment(href) or 'doc' in cls or 'nfw' in cls:
            name = a.get_text(strip=True) or '附件'
            full_url = urljoin(detail_url, href)
            if not any(a['url'] == full_url for a in attachments):
                attachments.append({'name': name, 'url': full_url})

    # 收集附件：匹配"附件："后面的链接
    attach_label = soup.find(string=re.compile(r'附件'))
    if attach_label:
        for sibling in attach_label.find_all_next('a'):
            href = sibling.get('href', '')
            if is_attachment(href):
                full_url = urljoin(detail_url, href)
                name = sibling.get_text(strip=True)
                if not any(a['url'] == full_url for a in attachments):
                    attachments.append({'name': name or '附件', 'url': full_url})

    # 正文提取
    editor = soup.select_one('div.TRS_Editor')
    if not editor:
        editor = soup.select_one('div.nyxqct')

    if not editor:
        content = ''
    else:
        # 收集附件链接，转为Markdown
        for a in editor.find_all('a'):
            href = a.get('href', '').strip()
            if not href:
                continue
            cls = ' '.join(a.get('class', [])) if a.get('class') else ''
            if is_attachment(href) or 'doc' in cls or 'nfw' in cls:
                name = a.get_text(strip=True) or '附件'
                full_url = urljoin(detail_url, href)
                a.replace_with(f' [{name}]({full_url}) ')

        # 剥离内联标签
        for tag in editor.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
            tag.unwrap()

        # 提取：表格保留HTML，段落取纯文本
        paragraphs = []
        for child in editor.children:
            if child.name in ('table',):
                html_str = str(child)
                # 清理表内多余属性，保留结构
                paragraphs.append(html_str)
            elif child.name == 'div':
                # 检查div内是否有table
                tbl = child.find('table')
                if tbl:
                    paragraphs.append(str(tbl))
                else:
                    text = child.get_text(separator='\n', strip=True)
                    if text:
                        paragraphs.append(text)
            elif child.name == 'p':
                text = child.get_text(separator='\n', strip=True)
                if text:
                    paragraphs.append(text)
            elif isinstance(child, str):
                text = child.strip()
                if text:
                    paragraphs.append(text)

        content = '\n\n'.join(paragraphs)

        # 空内容回退
        if len(content.strip()) < 20:
            content = ''

    # 空内容+仅附件
    if not content.strip() and attachments:
        content = f'[{full_title}]({detail_url})'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'
    elif content.strip() and attachments:
        # 有正文也有附件 → 在正文末尾追加附件链接
        content += '\n\n**附件：**'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'

    return content, attachments, full_title


def crawl_pages(start_page, end_page, cutoff, label):
    """爬取 start_page~end_page 范围内的所有文章"""
    all_items = []
    for p in range(start_page, end_page + 1):
        html = fetch_list_html(p)
        if not html:
            break
        items = parse_list(html)
        if not items:
            log(f"  [INFO] 第{p}页无数据，停止")
            break
        log(f"  第{p}页: {len(items)}条")
        all_items.extend(items)
        time.sleep(1)

    log(f"\n[INFO] 共 {len(all_items)} 条列表项，开始爬详情...")

    total = len(all_items)
    for idx, item in enumerate(all_items):
        # 日期过滤
        try:
            item_date = datetime.strptime(item['date'], '%Y-%m-%d')
            if item_date < cutoff:
                log(f"  [{idx+1}/{total}] ⏭️ 超出时间范围: {item['title'][:40]}")
                continue
        except ValueError:
            pass

        content, attachments, full_title = extract_content(item['url'], item['title'])
        if content is None:
            log(f"  [{idx+1}/{total}] ❌ {item['title'][:40]}")
            continue

        item['title'] = full_title  # 用完整标题覆盖截断的列表标题
        item['content'] = content
        item['attachments'] = attachments
        output_item(item, idx)

        if idx % 10 == 0:
            log(f"  [PROGRESS] {idx}/{total}")
        time.sleep(0.5)

    log(f"\n[DONE] 共处理 {total} 条")


def crawl_all(months_back=36):
    """全量爬取前5页（新站策略）"""
    cutoff = datetime.now() - timedelta(days=months_back * 30)
    end = min(TOTAL_PAGES, DEFAULT_FULL_PAGES)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 前{end}页 (新站策略, {months_back}个月)")
    log(f"{'='*50}")
    crawl_pages(1, end, cutoff, "full")


def crawl_all_pages():
    """爬取所有页（全量）"""
    cutoff = datetime.now() - timedelta(days=365 * 10)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 全量 {TOTAL_PAGES}页")
    log(f"{'='*50}")
    crawl_pages(1, TOTAL_PAGES, cutoff, "all")


def crawl_incremental():
    """增量：仅第1页"""
    cutoff = datetime.now() - timedelta(hours=48)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 增量（第1页）")
    log(f"{'='*50}")
    html = fetch_list_html(1)
    if not html:
        log("[ERROR] 无法获取第1页")
        return
    items = parse_list(html)
    log(f"第1页: {len(items)}条")

    total = len(items)
    for idx, item in enumerate(items):
        try:
            item_date = datetime.strptime(item['date'], '%Y-%m-%d')
            if item_date < cutoff:
                continue
        except ValueError:
            pass
        content, attachments, full_title = extract_content(item['url'], item['title'])
        if content is None:
            continue
        item['title'] = full_title
        item['content'] = content
        item['attachments'] = attachments
        output_item(item, idx)
        time.sleep(0.5)

    log(f"[DONE] 增量完成")


def output_item(item, idx):
    """输出单条记录为JSON行"""
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "attachments": item.get("attachments", []),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    print(json.dumps(record, ensure_ascii=False))


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        arg2 = sys.argv[2] if len(sys.argv) > 2 else ""
        if arg2 == "all":
            crawl_all_pages()
        else:
            months = int(arg2) if arg2 and arg2.isdigit() else 36
            crawl_all(months)
    elif mode == "list":
        html = fetch_list_html(1)
        items = parse_list(html)
        log(f"共 {len(items)} 条")
        for it in items[:5]:
            log(f"  {it['date']} | {it['title'][:50]} | {it['url']}")
    else:
        log(f"未知模式: {mode}")
        sys.exit(1)
