#!/usr/bin/env python3
"""
阳新县公示公告 (yx.gov.cn) 爬虫
https://www.yx.gov.cn/xwdt/gsgg/

CMS: 静态HTML分页 index_N.html
规则: 前 5 页，标题含"项目"，近 3 年
"""

import sys, os, re, json, time
from datetime import datetime, timezone, timedelta
import requests
import urllib3

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "阳新县公示公告"
BASE_URL = "https://www.yx.gov.cn"
LIST_PATH = "/xwdt/gsgg/index.html"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
}


def fetch_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ❌ 请求失败: {e}")
        return None


def parse_list(html):
    """解析列表页，返回 [(title, url, date), ...]"""
    items = []
    # <span class="date hidden-xs">2026-06-03</span><a href="..." title="xxx">display</a>
    pattern = r'<span[^>]*class="date[^"]*"[^>]*>(\d{4}-\d{2}-\d{2})</span>\s*<a href="([^"]+)"[^>]*title="([^"]*)"[^>]*>'
    for m in re.finditer(pattern, html):
        href = m.group(2)
        title = m.group(3).strip()
        date_str = m.group(1).strip()
        if not href or not title:
            continue
        if not href.startswith("http"):
            if href.startswith("./"):
                href = BASE_URL + "/xwdt/gsgg/" + href[2:]
            elif href.startswith("/"):
                href = BASE_URL + href
            else:
                href = BASE_URL + "/xwdt/gsgg/" + href
        items.append((title, href, date_str[:10]))
    return items


def fetch_detail(url):
    """获取详情页标题、正文、日期"""
    html = fetch_page(url)
    if not html:
        return None, None, None

    result = {}

    # 标题: <meta name="ArticleTitle" content="xxx"> 或 div.article h2
    m = re.search(r'<meta name="ArticleTitle"\s+content="([^"]*)"', html)
    if m:
        result["title"] = m.group(1).strip()
    else:
        m = re.search(r'<h2[^>]*>(.*?)</h2>', html, re.DOTALL)
        if m:
            result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()

    # 日期: <meta name="PubDate" content="2026-06-03">
    m = re.search(r'<meta name="PubDate"\s+content="(\d{4}-\d{1,2}-\d{1,2})"', html)
    if m:
        result["publish_date"] = m.group(1)
    else:
        m = re.search(r"发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})", html)
        if m:
            result["publish_date"] = m.group(1)

    # 正文: div.TRS_UEDITOR
    m = re.search(r'class="TRS_UEDITOR[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    else:
        m = re.search(r'class="view[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
        else:
            content = ""

    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL | re.I)
    result["content"] = content.strip()

    return (
        result.get("title"),
        result.get("content"),
        result.get("publish_date"),
    )


def main():
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME}")
    print(f"   前 {MAX_PAGES} 页, 标题含「项目」, 近3年")
    print(f"{'='*50}")

    # 收集所有列表项
    all_list_items = []
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = BASE_URL + LIST_PATH
        else:
            url = BASE_URL + f"/xwdt/gsgg/index_{page-1}.html"

        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch_page(url)
        if not html:
            print("❌ 无返回")
            break

        items = parse_list(html)
        if not items:
            print("0 条")
            break
        print(f"✅ {len(items)} 条")
        all_list_items.extend(items)

    print(f"\n{'─'*50}")
    print(f"📊 列表总计: {len(all_list_items)} 条")

    # 过滤 + 获取详情
    all_items = []
    seen_urls = set()

    for i, (title, url, date_str) in enumerate(all_list_items):
        if "项目" not in title:
            continue
        if date_str and date_str < THREE_YEARS_AGO:
            continue
        if url in seen_urls:
            continue
        seen_urls.add(url)

        print(f"  [{i+1}/{len(all_list_items)}] {title[:50]}...", end=" ", flush=True)
        detail_title, content, detail_date = fetch_detail(url)
        if detail_title:
            title = detail_title
        date = detail_date or date_str

        summary = re.sub(r"<[^>]+>", " ", content or "").strip()
        summary = re.sub(r"\s+", " ", summary)[:300]

        all_items.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": url,
            "content": content or "",
            "pub_date": date,
            "summary": summary,
            "tags": SITE_NAME,
        })
        print(f"✅")
        time.sleep(0.3)

    print(f"\n{'─'*50}")
    print(f"📊 总计匹配: {len(all_items)} 条")

    if not all_items:
        print("⏭️ 无数据，跳过推送")
        return

    print(f"\n📤 推送至服务器...")
    push_to_searchdb(all_items, "yx_gsgg")

    print(f"\n{'='*50}")
    print(f"✅ 完成! 共 {len(all_items)} 条已推送至服务器")
    print(f"{'='*50}")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time() - t0:.1f}s\n")
