#!/usr/bin/env python3
"""
印象庆阳网 - 公告/公示
URL: https://www.yinxiangqingyang.com/gonggao/
CMS: Cmstop (PHP)
Pagination: /gonggao/{n}.shtml (20 pages, ~50条/页)
List: div.article-list > h2.article-title a(标题+链接) + li.article-time(日期)
Detail: h1.title + div.article-content.article-content-show.fontSizeSmall
"""
import sys, os, re, json, time
from datetime import datetime, timedelta
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.yinxiangqingyang.com/gonggao/'
SITE = "印象庆阳网"
COLUMN = "公告公示"
PROVINCE = "企业"
TOTAL_PAGES = 20
DEFAULT_FULL_PAGES = 5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch_list_html(page):
    """Fetch list page HTML"""
    if page == 1:
        url = BASE_URL
    else:
        url = f'https://www.yinxiangqingyang.com/gonggao/{page}.shtml'
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            log(f"  [WARN] 第{page}页 status={r.status_code}")
            return None
        return r.text
    except Exception as e:
        log(f"  [ERROR] 第{page}页: {e}")
        return None


def parse_list(html):
    """Parse list items from HTML"""
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for article in soup.select('div.article-list'):
        h2 = article.select_one('h2.article-title a')
        time_el = article.select_one('li.article-time')
        if not h2 or not time_el:
            continue
        href = h2.get('href', '').strip()
        title = h2.get('title', '').strip() or h2.get_text(strip=True)
        date_str = time_el.get_text(strip=True)
        if not href or not title:
            continue
        full_url = urljoin(BASE_URL, href)
        # 日期格式: 2026/6/30 20:29:14 → 2026-06-30
        try:
            dt = datetime.strptime(date_str, '%Y/%m/%d %H:%M:%S')
            date_str = dt.strftime('%Y-%m-%d')
        except ValueError:
            pass
        items.append({'url': full_url, 'title': title, 'date': date_str})
    return items


def extract_content(detail_url, list_title):
    """提取正文和附件"""
    try:
        r = session.get(detail_url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            log(f"  [WARN] {detail_url} status={r.status_code}")
            return None, [], list_title
    except Exception as e:
        log(f"  [ERROR] {detail_url}: {e}")
        return None, [], list_title

    soup = BeautifulSoup(r.text, 'html.parser')

    # 完整标题
    h1 = soup.select_one('h1.title')
    full_title = h1.get_text(strip=True) if h1 else list_title

    # 收集附件
    attachments = []
    for a in soup.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', re.I)):
        href = a.get('href', '').strip()
        name = a.get_text(strip=True) or '附件'
        if href:
            full_url = urljoin(detail_url, href)
            attachments.append({'name': name, 'url': full_url})

    # 正文提取
    content_div = soup.select_one('div.article-content.article-content-show.fontSizeSmall')
    if not content_div:
        content_div = soup.select_one('div.article-content.fontSizeSmall')

    if not content_div:
        content = ''
    else:
        # 剥离内联标签和链接标签（防止内嵌URL导致文字被换行拆分）
        for tag in content_div.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's', 'a']):
            tag.unwrap()

        # 按 <p> 段落逐段提取，段落内用空连接符防止文字节点被\n拆分
        clean_lines = []
        for p in content_div.find_all('p'):
            text = p.get_text(separator='', strip=True)
            if text:
                clean_lines.append(text)
        content = '\n\n'.join(clean_lines)

        # 过滤末尾"编辑/XXX"行
        if content:
            last_newline = content.rfind('\n\n')
            if last_newline == -1:
                last_line = content
            else:
                last_line = content[last_newline+2:]
            if re.match(r'编辑[/／]', last_line.strip()):
                if last_newline == -1:
                    content = ''
                else:
                    content = content[:last_newline]

    # 正文末尾追加附件链接
    if content.strip() and attachments:
        content += '\n\n**附件：**'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'
    elif not content.strip() and attachments:
        content = f'<p><a href="{detail_url}">{full_title}</a></p>'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'

    return content, attachments, full_title


def crawl_pages(start_page, end_page, cutoff, label):
    """爬取 start_page~end_page"""
    all_items = []
    for p in range(start_page, end_page + 1):
        html = fetch_list_html(p)
        if not html:
            break
        items = parse_list(html)
        if not items:
            log(f"  [INFO] 第{p}页无数据，停止")
            break
        log(f"  第{p}页: {len(items)}条")
        all_items.extend(items)
        time.sleep(1)

    log(f"\n[INFO] 共 {len(all_items)} 条列表项，开始爬详情...")

    total = len(all_items)
    for idx, item in enumerate(all_items):
        # 日期过滤
        try:
            item_date = datetime.strptime(item['date'], '%Y-%m-%d')
            if item_date < cutoff:
                log(f"  [{idx+1}/{total}] ⏭️ 超出时间范围: {item['title'][:40]}")
                continue
        except ValueError:
            pass

        content, attachments, full_title = extract_content(item['url'], item['title'])
        if content is None:
            log(f"  [{idx+1}/{total}] ❌ {item['title'][:40]}")
            continue

        item['title'] = full_title
        item['content'] = content
        item['attachments'] = attachments
        output_item(item, idx)

        if idx % 10 == 0:
            log(f"  [PROGRESS] {idx}/{total}")
        time.sleep(0.3)

    log(f"\n[DONE] 共处理 {total} 条")


def crawl_all(months_back=36):
    """全量爬取前5页"""
    cutoff = datetime.now() - timedelta(days=months_back * 30)
    end = min(TOTAL_PAGES, DEFAULT_FULL_PAGES)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 前{end}页 (新站策略, {months_back}个月)")
    log(f"{'='*50}")
    crawl_pages(1, end, cutoff, "full")


def crawl_all_pages():
    """爬取所有20页"""
    cutoff = datetime.now() - timedelta(days=365 * 10)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 全量 {TOTAL_PAGES}页")
    log(f"{'='*50}")
    crawl_pages(1, TOTAL_PAGES, cutoff, "all")


def crawl_incremental():
    """增量：仅第1页"""
    cutoff = datetime.now() - timedelta(hours=48)
    log(f"\n{'='*50}")
    log(f"🏠 {SITE} - {COLUMN}")
    log(f"📄 增量（第1页）")
    log(f"{'='*50}")
    html = fetch_list_html(1)
    if not html:
        log("[ERROR] 无法获取第1页")
        return
    items = parse_list(html)
    log(f"第1页: {len(items)}条")

    total = len(items)
    for idx, item in enumerate(items):
        try:
            item_date = datetime.strptime(item['date'], '%Y-%m-%d')
            if item_date < cutoff:
                continue
        except ValueError:
            pass
        content, attachments, full_title = extract_content(item['url'], item['title'])
        if content is None:
            continue
        item['title'] = full_title
        item['content'] = content
        item['attachments'] = attachments
        output_item(item, idx)
        time.sleep(0.3)

    log(f"[DONE] 增量完成")


def output_item(item, idx):
    """输出JSON行"""
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "attachments": item.get("attachments", []),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    print(json.dumps(record, ensure_ascii=False))


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        arg2 = sys.argv[2] if len(sys.argv) > 2 else ""
        if arg2 == "all":
            crawl_all_pages()
        else:
            months = int(arg2) if arg2 and arg2.isdigit() else 36
            crawl_all(months)
    elif mode == "list":
        html = fetch_list_html(1)
        items = parse_list(html)
        log(f"共 {len(items)} 条")
        for it in items[:5]:
            log(f"  {it['date']} | {it['title'][:50]} | {it['url']}")
    else:
        log(f"未知模式: {mode}")
        sys.exit(1)
