#!/usr/bin/env python3
"""
甘南县 - 信息公开 (gannan.gov.cn)
列表: https://www.gannan.gov.cn/gannan/c100408/zfxxgk_list.shtml
翻页: index_2.shtml, index_3.shtml ...
详情: /gannan/c100408/202605/c02_623841.shtml
正文: article 区域或主内容 div
日期: meta PubDate 或 "发布时间"
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "甘南县-信息公开"
BASE_URL = "https://www.gannan.gov.cn"
LIST_URL = "https://www.gannan.gov.cn/gannan/c100408/zfxxgk_list.shtml"
LIST_DIR = "https://www.gannan.gov.cn/gannan/c100408/"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except:
        return None

def parse_list(html):
    """解析列表页，提取 <a> 链接 """
    items = []
    # 匹配 /gannan/c100408/.../...shtml 格式的链接
    pattern = r'<a[^>]*href="(/gannan/c100408/[^"]+\.shtml)"[^>]*>(.*?)</a>'
    for m in re.finditer(pattern, html, re.DOTALL):
        href = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        # 排除翻页链接、空标题、过短标题
        if not title or len(title) < 5:
            continue
        if 'index_' in href or 'zfxxgk_list' in href:
            continue
        items.append((title, BASE_URL + href))
    # 去重（按URL）
    seen = set()
    uniq = []
    for t, u in items:
        if u not in seen:
            seen.add(u)
            uniq.append((t, u))
    return uniq

def get_page_url(page):
    """第1页用主URL，第2页起用 index_N.shtml"""
    if page == 1:
        return LIST_URL
    return f"{LIST_DIR}index_{page}.shtml"

def fetch_detail(url):
    """提取详情页：标题、正文、发布日期"""
    html = fetch(url)
    if not html:
        return None, None, None

    result = {}

    # 标题：优先 h1，其次 article h2/h3
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        result["title"] = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    if not result.get("title"):
        m = re.search(r'<h2[^>]*>(.*?)</h2>', html, re.DOTALL)
        if m:
            result["title"] = re.sub(r'<[^>]+>', '', m.group(1)).strip()

    # 日期：meta PubDate
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{1,2}-\d{1,2})', html, re.I)
    if m:
        result["publish_date"] = m.group(1)
    if not result.get("publish_date"):
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m:
            result["publish_date"] = m.group(1)
    if not result.get("publish_date"):
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s*\d{2}:\d{2}', html)
        if m:
            result["publish_date"] = m.group(1)

    # 正文：尝试多种容器
    content = None
    # 优先 article 标签
    m = re.search(r'<article[^>]*>(.*?)</article>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()

    # 常见正文class
    if not content or len(content) < 100:
        for cls in ['article-content', 'article_content', 'content', 'article', 'text',
                     'main-content', 'mainContent', 'main', 'TRS_Editor', 'zoom',
                     'NewsContent', 'detail-content', 'detail_content']:
            m = re.search(
                r'<(div|section)[^>]*class="[^"]*' + cls + r'[^"]*"[^>]*>(.*?)</\1>',
                html, re.DOTALL
            )
            if m and len(m.group(2)) > 100:
                content = m.group(2).strip()
                break

    # 清洗
    if content and len(content) > 50:
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
        if len(content) > 100:
            result["content"] = content

    return result.get("title"), result.get("content"), result.get("publish_date")


def run(args=None):
    incremental = False
    if args and ('--incremental' in args or '-i' in args or '1' in args):
        incremental = True

    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")
    if incremental:
        print("📌 增量模式：仅爬取第1页")

    # --- 收集列表 ---
    all_list = []
    pages = 1 if incremental else MAX_PAGES
    for page in range(1, pages + 1):
        url = get_page_url(page)
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("❌ 请求失败")
            break
        items = parse_list(html)
        if not items:
            print("0 条")
            break
        print(f"✅ {len(items)} 条")
        all_list.extend(items)

    print(f"\n📊 列表总计: {len(all_list)} 条")

    # --- 抓取详情 ---
    all_items, seen = [], set()
    for i, (title, url) in enumerate(all_list):
        if url in seen:
            continue
        seen.add(url)

        print(f"  [{i+1}/{len(all_list)}] {title[:50]}...", end=" ", flush=True)

        dt, content, date = fetch_detail(url)
        if dt:
            title = dt

        # 日期过滤：超过3年的跳过
        if date and date < THREE_YEARS_AGO:
            print("⏭ 超时范围")
            continue

        if not content or len(content) < 50:
            print("⚠ 无正文")
            continue

        summary = re.sub(r'<[^>]+>', ' ', content).strip()[:300]
        summary = re.sub(r'\s+', ' ', summary)

        all_items.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": url,
            "content": content,
            "pub_date": date or "",
            "summary": summary,
            "tags": SITE_NAME,
        })
        print(f"✅ ({len(content)}字)")
        time.sleep(0.3)

    if all_items:
        push_to_searchdb(all_items, "gannan_gk")
    print(f"\n✅ 完成! 共入库 {len(all_items)} 条")


if __name__ == "__main__":
    t0 = time.time()
    run(sys.argv[1:])
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
