#!/usr/bin/env python3
"""瑞柏集团 - 新闻中心 爬虫
DCloud平台, JS渲染内容, 详情页正文为图片
"""
import re, requests
from bs4 import BeautifulSoup
from datetime import datetime
import sqlite3, os

BASE_URL = "https://www.ruibaigroup.com"
SITE_NAME = "ruibaigroup.com-新闻中心"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATEGORY = "eia"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
CUTOFF_DATE = "2023-06-16"


def get_conn():
    conn = sqlite3.connect(DB_PATH)
    conn.row_factory = sqlite3.Row
    return conn


def clean_text(text):
    if not text:
        return ""
    return re.sub(r'\s+', ' ', text.strip())


def extract_detail(session, article_id):
    """获取详情"""
    url = f"{BASE_URL}/news_detail/{article_id}.html"
    try:
        resp = session.get(url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [WARN] 获取详情页失败: {e}", flush=True)
        return None, None, None, None

    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')

    # 标题 - from <title> tag
    title_tag = soup.select_one('title')
    title = ''
    if title_tag:
        title = title_tag.get_text(strip=True).split('_')[0].strip()
    if not title:
        h1 = soup.select_one('h1')
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        return None, None, None, None

    # 日期
    pub_date = ''
    time_el = soup.select_one('.e_timeFormat-9, [class*=time]')
    if time_el:
        m = re.match(r'(\d{4}-\d{2}-\d{2})', time_el.get_text(strip=True))
        if m:
            pub_date = m.group(1)
    if not pub_date:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
        if m:
            pub_date = m.group(1)
    if not pub_date:
        return None, None, None, None

    # 正文 - e_richText-24 (DCloud平台), 图片转为base64内嵌
    content = ''
    content_el = soup.select_one('.e_richText-24')
    if content_el:
        content_html = str(content_el)
        # 下载图片并转为base64
        img_headers = {**HEADERS, 'Referer': BASE_URL + '/'}
        def embed_img(m):
            src = m.group(1)
            # 保留原始URL（不下载不转base64）
            return f'<img src=\"{src}\"' 
        content_html = re.sub(r'<img[^>]+?src=["\']([^"\']+)["\']', embed_img, content_html)
        content = content_html

    summary = ''
    if content:
        text = clean_text(content[:200])
        summary = text

    return title, pub_date, content, summary


def main():
    print(f"=== {SITE_NAME} 爬虫 ===", flush=True)
    print(f"数据库: {DB_PATH}", flush=True)

    conn = get_conn()
    cur = conn.cursor()
    cutoff_dt = datetime.strptime(CUTOFF_DATE, "%Y-%m-%d").date()

    session = requests.Session()

    # 分页获取所有文章 - DCloud标准格式
    articles = []
    seen_ids = set()
    for page in range(1, 30):
        url = f"{BASE_URL}/news_list/81163.html" if page == 1 else f"{BASE_URL}/news_list/81163-{page}-1.html"
        try:
            resp = session.get(url, headers=HEADERS, timeout=30)
            resp.encoding = "utf-8"
        except:
            continue
        
        # 跳过无文章数据的页
        if not re.search(r'/news_detail/\d+\.html', resp.text):
            if page > 10:
                break
            continue
        soup = BeautifulSoup(resp.text, 'html.parser')

        page_new = 0
        # 查找所有文章链接 - 兼容page1和分页页面的不同结构
        for a_tag in soup.select('a[href^="/news_detail/"]'):
            href = a_tag.get('href', '')
            m_id = re.search(r'/(\d+)\.html', href)
            if not m_id:
                continue
            aid = m_id.group(1)
            if aid in seen_ids:
                continue

            # 标题
            title = a_tag.get_text(strip=True)
            if not title or len(title) < 5:
                continue

            # 日期 - 从文章所在容器中查找时间元素
            container = a_tag.find_parent(['div', 'li'], class_=lambda c: c and 'loopitem' in str(c).lower() if c else False)
            if not container:
                container = a_tag.find_parent(['div', 'li'])
            pub_date = ''
            time_el = container.find_next(['p', 'span'], class_=lambda c: c and 'timeformat' in str(c).lower() if c else False) if container else None
            if not time_el:
                time_el = soup.find(['p', 'span'], class_=lambda c: c and 'timeformat' in str(c).lower() if c else False)
            if time_el:
                dm = re.match(r'(\d{4}-\d{2}-\d{2})', time_el.get_text(strip=True))
                if dm:
                    pub_date = dm.group(1)
            if not pub_date:
                dm = re.search(r'(\d{4}-\d{2}-\d{2})', resp.text)
                if dm:
                    pub_date = dm.group(1)

            seen_ids.add(aid)
            page_new += 1
            articles.append({
                'id': aid,
                'url': BASE_URL + href,
                'pub_date': pub_date,
            })

        if page_new == 0 and page > 20:
            print(f"  第{page}页: 无新内容, 停止分页", flush=True)
            break
        if page_new > 0:
            print(f"  第{page}页: {page_new}条新文章", flush=True)

    print(f"共 {len(articles)} 篇文章", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for art in articles:
        url = art['url']

        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if cur.fetchone():
            total_skip += 1
            continue

        pub_date = art['pub_date']
        if pub_date:
            pub_dt = datetime.strptime(pub_date, "%Y-%m-%d").date()
            if pub_dt < cutoff_dt:
                total_skip += 1
                continue

        print(f"  [{total_new+total_skip+total_error+1}/{len(articles)}] id={art['id']}", flush=True)
        title, detail_date, content, summary = extract_detail(session, art['id'])

        if not title:
            print(f"  [SKIP] 无详情", flush=True)
            total_error += 1
            continue

        date_rank = int(pub_date.replace('-', ''))
        source_url = f"{BASE_URL}/news_list/81163.html"

        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
                (url, title, content, SITE_NAME, pub_date, summary, CATEGORY, date_rank, source_url)
            )
            conn.commit()
            total_new += 1
            print(f"  [OK] {title[:40]}... | {pub_date}", flush=True)
        except Exception as e:
            print(f"  [ERROR] 入库失败: {title[:30]}... - {e}", flush=True)
            conn.rollback()
            total_error += 1

    conn.close()
    print(f"\n=== 完成: 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error} ===", flush=True)


if __name__ == '__main__':
    main()
