#!/usr/bin/env python3
"""
宜宾天原集团-信息公示 (ybty.com/xxgs/) 爬虫
https://www.ybty.com/xxgs/

CMS: 静态HTML分页 ?page=N
规则: 前 5 页
"""

import sys, os, re, time
from datetime import datetime
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "宜宾天原集团-信息公示"
BASE_URL  = "https://www.ybty.com"
LIST_URL  = "https://www.ybty.com/xxgs/"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
}

# 分页URL模式
PAGE_URL_PATTERN = "https://www.ybty.com/xxgs/?page={n}"


def fetch_page(url):
    """请求页面"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ❌ 请求失败: {e}")
        return None


def parse_list(html):
    """
    解析列表页，使用 BeautifulSoup
    返回: [(title, url, date_str), ...]
    
    结构:
        <a href="/xxgs/3110.html" class="linkb scs _in">
            <div class="txt">
                <time class="date">2026.04.20</time>
                <h2 class="title">标题文字</h2>
            </div>
        </a>
    """
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a_tag in soup.find_all("a", class_="linkb", href=True):
        href = a_tag.get("href", "").strip()
        if not href.startswith("/xxgs/") or not href.endswith(".html"):
            continue
        
        # 绝对URL
        full_url = urljoin(LIST_URL, href)
        
        # 标题
        title_elem = a_tag.find("h2", class_="title")
        title = title_elem.get_text(strip=True) if title_elem else ""
        
        # 日期
        time_elem = a_tag.find("time", class_="date")
        date_str = time_elem.get_text(strip=True) if time_elem else ""
        
        # 日期格式: YYYY.MM.DD -> YYYY-MM-DD
        if date_str:
            m = re.match(r'(\d{4})\.(\d{1,2})\.(\d{1,2})', date_str)
            if m:
                date_str = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"

        if title and full_url:
            items.append((title, full_url, date_str))
    
    return items


def fetch_detail(url):
    """
    获取详情页内容
    返回: {"content": "纯文本正文", "title": "标题"}
    
    标题: <title> 标签
    正文: <div class="sView-body _pr">...</div>
    """
    html = fetch_page(url)
    if not html:
        return {"content": "", "title": ""}

    soup = BeautifulSoup(html, "html.parser")

    # 标题: <title>标签
    title = ""
    title_tag = soup.find("title")
    if title_tag:
        title = title_tag.get_text(strip=True)
        # 去掉站点后缀
        title = re.sub(r'[-–—]宜宾天原集团股份有限公司$', '', title).strip()

    # 正文: <div class="sView-body _pr">
    content = ""
    body_div = soup.find("div", class_="sView-body _pr")
    if body_div:
        # 获取纯文本，保留基本的段落结构
        for tag in body_div.find_all(["script", "style"]):
            tag.decompose()
        content = body_div.get_text(separator="\n", strip=True)
    else:
        # 兜底：尝试找所有正文区域
        body_div = soup.find("div", class_="sView-body")
        if body_div:
            for tag in body_div.find_all(["script", "style"]):
                tag.decompose()
            content = body_div.get_text(separator="\n", strip=True)

    return {"content": content, "title": title}


def main():
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME}")
    print(f"   前 {MAX_PAGES} 页")
    print(f"{'='*50}")

    # 收集所有列表项
    all_list_items = []
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = PAGE_URL_PATTERN.replace("{n}", str(page))

        print(f"\n📄 第 {page} 页 ({url})...", end=" ", flush=True)
        html = fetch_page(url)
        if not html:
            print("❌ 无返回")
            break

        items = parse_list(html)
        if not items:
            print("0 条")
            break
        print(f"✅ {len(items)} 条")
        all_list_items.extend(items)

    print(f"\n{'─'*50}")
    print(f"📊 列表总计: {len(all_list_items)} 条")

    # 去重（按URL）
    seen_urls = set()
    unique_items = []
    for title, url, date_str in all_list_items:
        if url not in seen_urls:
            seen_urls.add(url)
            unique_items.append((title, url, date_str))
    
    print(f"📊 去重后: {len(unique_items)} 条")

    # 爬详情
    all_entries = []
    total = len(unique_items)
    for i, (title, url, date_str) in enumerate(unique_items, 1):
        print(f"  [{i}/{total}] {title[:50]}...", end=" ", flush=True)
        
        detail = fetch_detail(url)
        detail_title = detail.get("title", "")
        content = detail.get("content", "")
        
        # 优先用详情页标题，其次用列表页标题
        final_title = detail_title or title
        
        # 清理正文中多余空白
        if content:
            content = re.sub(r'\n{3,}', '\n\n', content)
        
        entry = {
            "site_name": SITE_NAME,
            "title": final_title,
            "url": url,
            "content": content,
            "pub_date": date_str,
            "summary": (final_title[:200] if final_title else ""),
            "tags": SITE_NAME,
        }
        all_entries.append(entry)
        print(f"✅ [{date_str}]")
        time.sleep(0.5)

    print(f"\n{'─'*50}")
    print(f"📊 总计: {len(all_entries)} 条待推送")

    if not all_entries:
        print("⏭️ 无数据，跳过推送")
        return

    print(f"\n📤 推送至服务器...")
    push_to_searchdb(all_entries, "ybty_xxgs")

    print(f"\n{'='*50}")
    print(f"✅ 完成! 共 {len(all_entries)} 条已推送至服务器")
    print(f"{'='*50}")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time() - t0:.1f}s\n")
