#!/usr/bin/env python3
"""
万年县通知公告 (zgwn.gov.cn) 爬虫
http://www.zgwn.gov.cn/zgwn/tzgg/list_common_list.shtml

CMS: UCAP CMS, 静态HTML分页
规则: 前 5 页，标题含"项目"，近 3 年
"""

import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests
import urllib3

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "万年县通知公告"
BASE_URL = "http://www.zgwn.gov.cn"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
}


def fetch_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ❌ 请求失败: {e}")
        return None


def parse_list(html):
    """解析列表页，返回 [(title, url, date), ...]"""
    items = []
    # <li><a href="/zgwn/tzgg/202605/xxx.shtml">Title 2026-05-27</a></li>
    # Some links point to external sites (mp.weixin.qq.com) - skip those
    pattern = r'<li>.*?<a href="(/zgwn/[^"]+)"[^>]*>(.*?)</a>'
    for m in re.finditer(pattern, html, re.DOTALL):
        href = m.group(1)
        text = m.group(2).strip()
        # Skip if not local zgwn link
        if not href.startswith("/zgwn/"):
            continue
        # Extract title and date from text
        # Format: "Title 2026-05-27"
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})\s*$', text)
        date_str = date_m.group(1) if date_m else ""
        title = re.sub(r'\s+\d{4}-\d{2}-\d{2}\s*$', '', text).strip()
        if not title:
            continue
        full_url = BASE_URL + href
        items.append((title, full_url, date_str))
    return items


def fetch_detail(url):
    """获取详情页标题、正文、日期"""
    html = fetch_page(url)
    if not html:
        return None, None, None

    result = {}

    # 标题: <h1>xxx</h1>
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()

    # 日期: <meta name="PubDate" content="2026-03-31"> 或 content="2026-03-31 15:44:36"
    m = re.search(r'<meta name="PubDate"\s+content="(\d{4}-\d{1,2}-\d{1,2})', html)
    if m:
        result["publish_date"] = m.group(1)

    # 正文: <div class="wzcon ...">...</div>
    m = re.search(r'<div[^>]*class="wzcon[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL | re.I)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL | re.I)
        result["content"] = content

    return (
        result.get("title"),
        result.get("content"),
        result.get("publish_date"),
    )


def get_pagination_info(html):
    """获取总页数"""
    # Look for createPageHTML(N, 0, ...) pattern
    m = re.search(r'createPageHTML\((\d+),', html)
    if m:
        return int(m.group(1))
    return 1


def main():
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME}")
    print(f"   前 {MAX_PAGES} 页, 标题含「项目」, 近3年")
    print(f"{'='*50}")

    all_list_items = []
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = f"{BASE_URL}/zgwn/tzgg/list_common_list.shtml"
        else:
            url = f"{BASE_URL}/zgwn/tzgg/list_common_list_{page}.shtml"

        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch_page(url)
        if not html:
            print("❌ 无返回")
            break

        items = parse_list(html)
        if not items:
            print("0 条")
            break
        print(f"✅ {len(items)} 条")
        all_list_items.extend(items)

    print(f"\n{'─'*50}")
    print(f"📊 列表总计: {len(all_list_items)} 条")

    all_items = []
    seen_urls = set()

    for i, (title, url, date_str) in enumerate(all_list_items):
        if "项目" not in title:
            continue
        if date_str and date_str < THREE_YEARS_AGO:
            continue
        if url in seen_urls:
            continue
        seen_urls.add(url)

        print(f"  [{i+1}/{len(all_list_items)}] {title[:50]}...", end=" ", flush=True)
        detail_title, content, detail_date = fetch_detail(url)
        if detail_title:
            title = detail_title
        date = detail_date or date_str

        summary = re.sub(r"<[^>]+>", " ", content or "").strip()
        summary = re.sub(r"\s+", " ", summary)[:300]

        all_items.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": url,
            "content": content or "",
            "pub_date": date,
            "summary": summary,
            "tags": SITE_NAME,
        })
        print(f"✅")
        time.sleep(0.3)

    print(f"\n{'─'*50}")
    print(f"📊 总计匹配: {len(all_items)} 条")

    if not all_items:
        print("⏭️ 无数据，跳过推送")
        return

    print(f"\n📤 推送至服务器...")
    push_to_searchdb(all_items, "zgwn_tzgg")

    print(f"\n{'='*50}")
    print(f"✅ 完成! 共 {len(all_items)} 条已推送至服务器")
    print(f"{'='*50}")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time() - t0:.1f}s\n")
