#!/usr/bin/env python3
"""安徽昊源化工 - 环境保护 爬虫
站点: www.chinahaoyuan.com (企业站)
栏目: 环境保护 (/lingdaoweiwen.html)
列表: /lingdaoweiwen.html 分页 /lingdaoweiwen/{n}.html (共19页)
详情: /lingdaoweiwen/detail/{id}.html
标题: div.view-title
日期: 发布时间：YYYY-MM-DD
正文: div.n_content_c
"""

import sys, os, re, time
sys.path.insert(0, "/root/gov_crawler")
from base_crawler import GovCrawler

SITE_NAME = "昊源化工-环境保护"
DOMAIN = "www.chinahaoyuan.com"
BASE_URL = "https://www.chinahaoyuan.com"


class HaoyuanEnvCrawler(GovCrawler):
    def do_crawl(self):
        total_pages = getattr(self, 'max_pages', 19)
        for page in range(1, total_pages + 1):
            if page == 1:
                url = f"{BASE_URL}/lingdaoweiwen.html"
            else:
                url = f"{BASE_URL}/lingdaoweiwen/{page}.html"

            html = self.http_get(url)
            if not html:
                print(f"  ⚠️ 第{page}页获取失败")
                continue

            from bs4 import BeautifulSoup
            soup = BeautifulSoup(html, 'html.parser')

            items_found = 0
            # 列表项: <li><a href="/lingdaoweiwen/detail/XXXX.html">
            for li in soup.find_all('li'):
                a = li.find('a', href=re.compile(r'/lingdaoweiwen/detail/\d+\.html'))
                if not a:
                    continue

                href = a.get('href', '')
                detail_url = BASE_URL + href

                # 标题: 优先取 title 属性，其次取 h3 内容
                title = (a.get('title') or '').strip()
                if not title:
                    h3 = a.find('h3')
                    title = h3.get_text(strip=True) if h3 else ''

                # 日期: <span>YYYY-MM-DD</span>
                pub_date = ''
                span = a.find('span')
                if span:
                    date_m = re.search(r'(\d{4}-\d{2}-\d{2})', span.get_text(strip=True))
                    if date_m:
                        pub_date = date_m.group(1)

                if not title:
                    continue

                # 获取详情
                detail_html = self.http_get(detail_url)
                if not detail_html:
                    print(f"    ⚠️ 详情页获取失败: {detail_url}")
                    continue

                # 提取详情
                detail_soup = BeautifulSoup(detail_html, 'html.parser')

                # 标题: 详情页的 view-title 优先
                view_title = detail_soup.find('div', class_='view-title')
                detail_title = view_title.get_text(strip=True) if view_title else title

                # 日期
                detail_date = pub_date
                date_el = detail_soup.find('div', class_='view-element')
                if date_el:
                    dm = re.search(r'(\d{4}-\d{2}-\d{2})', date_el.get_text())
                    if dm:
                        detail_date = dm.group(1)

                # 正文: div.n_content_c
                content_div = detail_soup.find('div', class_='n_content_c')
                content = ''
                if content_div:
                    content = str(content_div)

                summary = ''
                if content_div:
                    summary = content_div.get_text(strip=True)[:500]

                # 附件: 百度网盘链接
                attachments = []
                for a_tag in detail_soup.find_all('a', href=re.compile(r'pan\.baidu\.com')):
                    attachments.append({
                        "name": a_tag.get_text(strip=True) or "百度网盘链接",
                        "url": a_tag.get('href', '')
                    })

                # 存储
                self.store_item(
                    title=detail_title,
                    url=detail_url,
                    content=content,
                    date=detail_date,
                    summary=summary,
                )
                items_found += 1
                time.sleep(self.sync_delay)

            print(f"  📄 第{page}页: {items_found}条")
            time.sleep(self.sync_delay * 2)

        print(f"\n  ✅ 完成: 新增{self._stats['new']}, 跳过{self._stats['skip']}, 错误{self._stats['errors']}")


if __name__ == '__main__':
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=19, help='页数')
    args = parser.parse_args()
    c = HaoyuanEnvCrawler(
        site_name=SITE_NAME,
        domain=DOMAIN,
        url=BASE_URL,
        sync_delay=0.3,
    )
    c.max_pages = args.pages
    c.run()
