#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫：润和催化剂股份有限公司 - 公示（环评/环境影响评价公告）
站点：www.rezel.com.cn
CMS：动力视线 dlssyht.cn (易维)
"""

import sys, os, re, requests
from bs4 import BeautifulSoup

_HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, _HERE)
from crawler_lib import push_to_searchdb

SITE_NAME = "润和催化剂公告"

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'
}

BASE_URL = 'http://www.rezel.com.cn'
LIST_URL = 'http://www.rezel.com.cn/doc_28078210_6026322_0_1.html'

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list():
    r = session.get(LIST_URL, timeout=30)
    r.encoding = 'gbk'
    soup = BeautifulSoup(r.text, 'html.parser')
    items = []
    for span_a in soup.select('span.text-list-a'):
        a_tag = span_a.find('a')
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)
        if not title:
            continue
        date_span = span_a.find_next('span', class_='text-list-times')
        date_str = date_span.get_text(strip=True) if date_span else ''
        if href and not href.startswith('http'):
            href = BASE_URL + href
        items.append({'title': title, 'url': href, 'date': date_str})
    return items


def fetch_detail(url):
    r = session.get(url, timeout=30)
    r.encoding = 'gbk'
    soup = BeautifulSoup(r.text, 'html.parser')
    content_div = soup.find('div', class_=re.compile(r'hb-body-inner'))
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'ev-text-article'))
    if not content_div:
        return ''
    
    # 提取所有 <p> 段落，跳过含子 <p> 的包装层
    parts = []
    for p in content_div.find_all('p'):
        if p.find('p'):
            continue  # 跳过包装层 <p>
        text = p.get_text(' ', strip=True)
        if text:
            parts.append(text)
    
    if not parts:
        text = content_div.get_text('\n', strip=True)
        if text:
            parts = [text]
    
    body = '\n\n'.join(parts)
    body = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])', '', body)
    return body


def main():
    print(f"[{SITE_NAME}] 开始爬取")
    items = fetch_list()
    print(f"  列表获取 {len(items)} 条")
    if not items:
        print("  列表为空，退出")
        return

    db_items = []
    for item in items:
        url = item['url']
        title = item['title']
        date_str = item['date']
        print(f"  详情: {title[:40]}...")
        body = fetch_detail(url)
        if not body:
            print(f"    ⚠️ 正文为空")
        
        db_items.append({
            'site_name': SITE_NAME,
            'source_url': url,
            'url': url,
            'title': title,
            'pub_date': date_str,
            'summary': body[:500] if body else '',
            'content': body,
        })
    
    push_to_searchdb(db_items, batch_label=SITE_NAME)
    print(f"\n[{SITE_NAME}] 完成: 共 {len(items)} 条")


if __name__ == '__main__':
    main()
