#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫：湖北省宏源药业科技股份有限公司 - 临时公告（环评公示）
站点：www.hbhypharm.com
CMS：中企动力 dcloud (300.cn)
"""

import sys, os, re, requests
from bs4 import BeautifulSoup

_HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, _HERE)
from crawler_lib import push_to_searchdb

SITE_NAME = "宏源药业-临时公告"

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'
}

BASE_URL = 'https://www.hbhypharm.com'
PAGE_ID = '1729775582547165184'
PAGE_SIZE = 8

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list_page(page_url):
    """获取列表页"""
    r = session.get(page_url, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    
    items = []
    # 第二个 p_list 才是新闻列表
    p_lists = soup.find_all('div', class_='p_list')
    if len(p_lists) < 2:
        return items, False
    
    news_list = p_lists[1]
    loop_items = news_list.find_all('div', class_=re.compile(r'p_loopitem'))
    
    for loop_item in loop_items:
        a = loop_item.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not title:
            continue
        if href and not href.startswith('http'):
            href = BASE_URL + href
        
        # 日期从 e_timeFormat-16 提取
        time16 = loop_item.find('p', class_=re.compile(r'e_timeFormat-16'))
        date_str = ''
        if time16:
            text = time16.get_text(strip=True)
            m = re.search(r'(\d{4}-\d{2}-\d{2})', text)
            if m:
                date_str = m.group(1)
        
        items.append({'title': title, 'url': href, 'date': date_str})
    
    return items, True


def fetch_detail(url):
    """获取详情页正文"""
    r = session.get(url, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # 正文在 div.e_richText 内
    content_div = soup.find('div', class_=re.compile(r'e_richText'))
    if not content_div:
        # 备选
        for cls in ['e_text', 'p_showText', 'showText']:
            content_div = soup.find('div', class_=re.compile(cls))
            if content_div:
                break
    
    if not content_div:
        return ''
    
    # 提取 <p> 段落
    parts = []
    for p in content_div.find_all('p'):
        text = p.get_text(' ', strip=True)
        if text:
            parts.append(text)
    
    if not parts:
        text = content_div.get_text('\n', strip=True)
        if text:
            parts = [text]
    
    body = '\n\n'.join(parts)
    # 清理中文字间多余空格
    body = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])', '', body)
    return body


def main():
    # 计算总页数
    total_pages = 2  # 已知：page1 8条, page2 6条
    
    all_items = []
    
    # Page 1: /news/10/
    page1_url = f'{BASE_URL}/news/10/'
    print(f"[{SITE_NAME}] 爬取第1页: {page1_url}")
    items, _ = fetch_list_page(page1_url)
    print(f"  获取 {len(items)} 条")
    all_items.extend(items)
    
    # Page 2+: news_list/{pageId}-{offset}-{pageSize}.html
    for page_num in range(2, total_pages + 1):
        offset = (page_num - 1) * PAGE_SIZE
        page_url = f'{BASE_URL}/news_list/{PAGE_ID}-{offset}-{PAGE_SIZE}.html'
        print(f"  爬取第{page_num}页: {page_url}")
        items, has_more = fetch_list_page(page_url)
        print(f"  获取 {len(items)} 条")
        all_items.extend(items)
        if not has_more:
            break
    
    print(f"\n列表合计: {len(all_items)} 条")
    
    if not all_items:
        print("  列表为空，退出")
        return
    
    db_items = []
    for item in all_items:
        url = item['url']
        title = item['title']
        date_str = item['date']
        print(f"  详情: {title[:40]}...")
        body = fetch_detail(url)
        if not body:
            print(f"    ⚠️ 正文为空")
        
        db_items.append({
            'site_name': SITE_NAME,
            'source_url': url,
            'url': url,
            'title': title,
            'pub_date': date_str,
            'summary': body[:500] if body else '',
            'content': body,
        })
    
    push_to_searchdb(db_items, batch_label=SITE_NAME)
    print(f"\n[{SITE_NAME}] 完成: 共 {len(all_items)} 条")


if __name__ == '__main__':
    main()
