#!/usr/bin/env python3
"""
crawl_zuoyun.py — 左云县人民政府·公示公告 爬虫
=============================================
列表：所有条目一次性加载在 HTML 中（客户端分页，无需翻页）
详情：<div class='TRS_Editor'><ucapcontent>...</ucapcontent></div>
日期：<meta name="PubDate" content="YYYY-MM-DD HH:mm"/>

用法:
    python3 crawl_zuoyun.py             # 全量爬（近3年，含"项目"）
    python3 crawl_zuoyun.py --limit=5   # 测试只跑5条
    python3 crawl_zuoyun.py --stats     # 看统计
"""

import re, sys, os, time
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import (
    fetch_page, parse_date, clean_html, THREE_YEARS_AGO,
    extract_date_from_html, extract_content_by_selector, extract_content_by_regex,
    save_jsonl, push_to_searchdb
)

# ═══════════════════════════════════════════
#  配置区
# ═══════════════════════════════════════════

SITE_NAME = "左云县人民政府"
BASE_URL  = "https://www.zuoyun.gov.cn"
LIST_URL  = "https://www.zuoyun.gov.cn/zyxrmzfz/gsgg/list.shtml"

# 列表页：所有条目在 <ul class="list-cj-gl xmt_none"> 中
# <li data-time="2026-06-02" data-url="../../zyxrmzfz/gsgg/202606/xxx.shtml" data-title="标题">
LIST_ITEM_PATTERN = r'<li[^>]*data-time="([^"]*)"[^>]*data-url="([^"]*)"[^>]*data-title="([^"]*)"[^>]*>'

# 详情页正文 CSS 选择器
DETAIL_SELECTOR = ".TRS_Editor"

# 日期元数据
META_PUBDATE = re.compile(
    r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)"', re.I
)


def _abs_urls(html, base_url):
    """将 HTML 中的相对路径（src/href）转为绝对 URL"""
    def _abs(m):
        attr = m.group(1)
        val = m.group(2)
        if val.startswith(('http://', 'https://', '//', 'data:', 'javascript:', '#', 'tel:', 'mailto:')):
            return m.group(0)
        abs_val = urljoin(base_url, val)
        return f'{attr}="{abs_val}"'
    html = re.sub(r'(src|href)="([^"]+)"', _abs, html, flags=re.I)
    return html


# ═══════════════════════════════════════════
#  列表页提取
# ═══════════════════════════════════════════

def get_list_items(html):
    """从列表页提取所有条目（全部在 HTML 中）"""
    items = []
    for m in re.finditer(LIST_ITEM_PATTERN, html):
        date_str = m.group(1).strip()
        href = m.group(2).strip()
        title = m.group(3).strip()
        if not title or not href:
            continue
        full_url = urljoin(LIST_URL, href)
        items.append({
            "title": title,
            "url": full_url,
            "pub_date": date_str,
        })
    return items


# ═══════════════════════════════════════════
#  详情页提取
# ═══════════════════════════════════════════

def get_detail(url):
    """访问详情页，返回正文 HTML 和发布时间"""
    html = fetch_page(url)
    if not html:
        return {"content": "", "pub_date": ""}

    # 1. 尝试 CSS 选择器 .TRS_Editor
    content = extract_content_by_selector(html, DETAIL_SELECTOR)

    # 2. 提取 <ucapcontent> 内部
    if content and '<ucapcontent>' in content:
        m = re.search(r'<ucapcontent>(.*?)</ucapcontent>', content, re.DOTALL)
        if m:
            content = clean_html(m.group(1))
        else:
            content = clean_html(content)

    # 3. 手动正则兜底
    if not content or len(content) < 50:
        m = re.search(r'<ucapcontent>(.*?)</ucapcontent>', html, re.DOTALL)
        if m:
            content = clean_html(m.group(1))

    # 4. 相对路径 → 绝对路径（图片、附件链接等）
    if content:
        content = _abs_urls(content, url)

    # 5. 日期
    pub_date = ""
    m = META_PUBDATE.search(html)
    if m:
        pub_date = parse_date(m.group(1))

    return {"content": content, "pub_date": pub_date}


# ═══════════════════════════════════════════
#  主流程
# ═══════════════════════════════════════════

def crawl(max_items=None):
    """全量爬取（近3年，含"项目"）"""
    print(f"  📡 请求列表页: {LIST_URL}")
    html = fetch_page(LIST_URL)
    if not html:
        print("  ❌ 列表页请求失败")
        return 0, 0

    all_items = get_list_items(html)
    print(f"  列表页: 共 {len(all_items)} 条")

    # 过滤：标题含"项目" + 近3年
    filtered = []
    for it in all_items:
        if "项目" not in it["title"]:
            continue
        if it.get("pub_date") and it["pub_date"] < THREE_YEARS_AGO:
            continue
        filtered.append(it)

    # 去重
    seen = set()
    unique = []
    for it in filtered:
        if it["url"] not in seen:
            seen.add(it["url"])
            unique.append(it)

    print(f"  过滤后: {len(unique)} 条（含\"项目\"+近3年）")

    if max_items:
        unique = unique[:max_items]

    # 爬正文
    collected = []
    ok, fail = 0, 0
    for i, item in enumerate(unique, 1):
        detail = get_detail(item["url"])
        entry = {
            "site_name": SITE_NAME,
            "title": item["title"],
            "url": item["url"],
            "content": detail["content"],
            "pub_date": detail["pub_date"] or item["pub_date"],
            "tags": SITE_NAME,
        }
        collected.append(entry)
        save_jsonl(entry)
        ok += 1
        if i % 5 == 0 or i == len(unique):
            print(f"  [{i}/{len(unique)}] ✅{ok} ❌{fail}")

    # 推送到服务器 search.db
    if collected:
        push_to_searchdb(collected, batch_label=SITE_NAME.replace(" ", "_"))
    print(f"  ✅ 完成: 共{ok}条")
    return ok, fail


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description=f"爬虫: {SITE_NAME}")
    parser.add_argument("--limit", type=int, help="测试：只跑 N 条")
    parser.add_argument("--stats", action="store_true", help="查看 eia.db 统计")
    args = parser.parse_args()

    if args.stats:
        print("📊 统计数据: 请登录服务器查看 search.db")
        sys.exit(0)

    print(f"\n📡 [{SITE_NAME}] 开始爬取 (含\"项目\"+近3年)")
    t0 = time.time()
    crawl(max_items=args.limit)
    print(f"⏱ 耗时: {time.time()-t0:.1f}s\n")
