#!/usr/bin/env python3
"""
垣曲县生态环境 (yuanqu.gov.cn) 爬虫
http://www.yuanqu.gov.cn/zfxxgk/jczwgk/gknr/gknr_hjbh/index.shtml

CMS: 静态HTML，分页 index_N.shtml
规则: 全部9页，标题含"项目"，近3年
"""

import sys, os, re, json, time
from datetime import datetime, timezone, timedelta
import requests
import urllib3

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "垣曲县生态环境"
BASE_URL = "http://www.yuanqu.gov.cn"
LIST_URL = "/zfxxgk/jczwgk/gknr/gknr_hjbh/index.shtml"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365 * 3)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
}


def fetch_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ❌ 请求失败: {e}")
        return None


def parse_list(html):
    """解析列表页，返回 [(title, url, date), ...]"""
    items = []
    # 匹配 <li><a href="..." title="full title">display...</a><span>2026-05-12</span></li>
    pattern = r'<li><a href="([^"]+)"[^>]*title="([^"]*)"[^>]*>.*?</a><span>([^<]+)</span></li>'
    for m in re.finditer(pattern, html):
        href = m.group(1)
        title = m.group(2).strip()
        date_str = m.group(3).strip()
        if not href or not title:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append((title, href, date_str[:10]))
    return items


def get_total_pages(html):
    """获取总页数"""
    m = re.search(r'"pageCount":"(\d+)"', html)
    return int(m.group(1)) if m else 1


def fetch_detail(url):
    """获取详情页标题、正文、日期"""
    html = fetch_page(url)
    if not html:
        return None, None, None

    result = {}

    # 标题: <h1 class="green">xxx</h1>
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()

    # 日期
    m = re.search(r"发布日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})", html)
    if m:
        result["publish_date"] = m.group(1)

    # 正文: <div id="Zoom">...</div>
    m = re.search(r'<div id="Zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        # 清理: 去掉 script/style
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL | re.I)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL | re.I)
        result["content"] = content

    return (
        result.get("title"),
        result.get("content"),
        result.get("publish_date"),
    )


def main(incremental=False):
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME}")
    print(f"   爬取全部页面，标题含「项目」、近3年")
    print(f"{'='*50}")

    # 获取第一页，确定总页数
    html = fetch_page(BASE_URL + LIST_URL)
    if not html:
        print("❌ 无法获取列表页")
        return

    total_pages = get_total_pages(html)
    print(f"📊 共 {total_pages} 页")

    # 收集列表项
    all_list_items = []
    for page in range(1, 2 if incremental else total_pages + 1):
        if page == 1:
            url = BASE_URL + LIST_URL
        else:
            url = BASE_URL + re.sub(r"/index\.", f"/index_{page}.", LIST_URL)

        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch_page(url)
        if not html:
            print("❌ 无返回")
            break

        items = parse_list(html)
        print(f"✅ {len(items)} 条")
        all_list_items.extend(items)

    print(f"\n{'─'*50}")
    print(f"📊 列表总计: {len(all_list_items)} 条")

    # 过滤 + 获取详情
    all_items = []
    seen_urls = set()

    for i, (title, url, date_str) in enumerate(all_list_items):
        # 过滤: 标题含"项目"
        if "项目" not in title:
            continue

        # 过滤: 近 3 年
        if date_str and date_str < THREE_YEARS_AGO:
            continue

        if url in seen_urls:
            continue
        seen_urls.add(url)

        print(f"  [{i+1}/{len(all_list_items)}] {title[:50]}...", end=" ", flush=True)
        detail_title, content, detail_date = fetch_detail(url)
        if detail_title:
            title = detail_title
        date = detail_date or date_str

        summary = re.sub(r"<[^>]+>", " ", content or "").strip()
        summary = re.sub(r"\s+", " ", summary)[:300]

        all_items.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": url,
            "content": content or "",
            "pub_date": date,
            "summary": summary,
            "tags": SITE_NAME,
        })
        print(f"✅")
        time.sleep(0.3)

    print(f"\n{'─'*50}")
    print(f"📊 总计匹配: {len(all_items)} 条")

    if not all_items:
        print("⏭️ 无数据，跳过推送")
        return

    print(f"\n📤 推送至服务器...")
    push_to_searchdb(all_items, "yuanqu_sthj")

    print(f"\n{'='*50}")
    print(f"✅ 完成! 共 {len(all_items)} 条已推送至服务器")
    print(f"{'='*50}")


if __name__ == "__main__":
    t0 = time.time()
    main(incremental=len(sys.argv) > 1 and sys.argv[1] in ("1", "--incremental"))
    print(f"⏱ 耗时: {time.time() - t0:.1f}s\n")
