#!/usr/bin/env python3
"""建峰集团 - 环境公示 爬虫
站点: www.cnjf.com (企业站)
栏目: 信息公开 > 环境公示 (classid=26)
列表: /aspx/ch/dutylist.aspx?classid=26&page=N  (N=1..4)
详情: /aspx/ch/show.aspx?classid=26&id=XXXX
标题: <h3> 标签
日期: 发布时间： YYYY-MM-DD
"""

import sys, os, re, time
sys.path.insert(0, "/root/gov_crawler")
from base_crawler import GovCrawler

SITE_NAME = "建峰集团-环境公示"
DOMAIN = "www.cnjf.com"
BASE_URL = "https://www.cnjf.com"
LIST_TPL = "https://www.cnjf.com/aspx/ch/dutylist.aspx?classid=26&page={}"


class CnjfHjgsCrawler(GovCrawler):
    def do_crawl(self):
        total_pages = getattr(self, 'max_pages', 4)
        for page in range(1, total_pages + 1):
            url = LIST_TPL.format(page)
            html = self.http_get(url)
            if not html:
                print(f"  ⚠️ 第{page}页获取失败")
                continue

            from bs4 import BeautifulSoup
            soup = BeautifulSoup(html, 'html.parser')

            # 提取列表项: <a href="show.aspx?classid=26&id=XXXX">YYYY MM-DD 标题</a>
            # 页面1: 链接文本直接含日期前缀
            # 页面2-4: 结构为 <li><a><p>YYYY</p>MM-DD<p>标题</p></a></li>
            items_found = 0
            for a in soup.find_all('a', href=re.compile(r'show\.aspx\?')):
                href = a.get('href', '')
                # id= 参数在 &id= 或 ?id= 后面（不是 classid=）
                id_m = re.search(r'[?&]id=(\d+)', href)
                if not id_m:
                    continue
                detail_id = id_m.group(1)
                # 标题: 去掉 "YYYYMM-DD " 或 "YYYY MM-DD " 前缀
                title = a.get_text(strip=True)
                clean_title = re.sub(r'^\d{4}\s?\d{2}-\d{2}\s+', '', title).strip()
                if not clean_title:
                    clean_title = title
                classid_m = re.search(r'classid=(\d+)', href)
                detail_url = f"{BASE_URL}/aspx/ch/show.aspx?classid={classid_m.group(1) if classid_m else '26'}&id={detail_id}"

                # 获取详情
                detail_html = self.http_get(detail_url)
                if not detail_html:
                    print(f"    ⚠️ 详情页获取失败: {detail_id}")
                    continue

                # 提取标题、日期、正文
                title_from_detail, content, pub_date, attachments, summary = self._extract_detail(
                    detail_html, clean_title)

                if not title_from_detail:
                    print(f"    ⚠️ 跳过无标题: id={detail_id}")
                    continue

                # 存储
                self.store_item(
                    title=title_from_detail,
                    url=detail_url,
                    content=content,
                    date=pub_date,
                    summary=summary,
                )
                items_found += 1
                time.sleep(self.sync_delay)

            print(f"  📄 第{page}页: {items_found}条")
            time.sleep(self.sync_delay)

        print(f"\n  ✅ 完成: 新增{self._stats['new']}, 跳过{self._stats['skip']}, 错误{self._stats['errors']}")

    def _extract_detail(self, html, fallback_title):
        """从详情页提取标题、正文、日期"""
        from bs4 import BeautifulSoup
        soup = BeautifulSoup(html, 'html.parser')

        # 标题: <h3>
        h3 = soup.find('h3')
        title = h3.get_text(strip=True) if h3 else fallback_title

        # 日期
        pub_date = ""
        date_m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if date_m:
            pub_date = date_m.group(1)

        # 正文: h3 之后到 "上一条"/"下一条" 之间的所有文本
        content_parts = []
        if h3:
            current = h3.find_next_sibling()
            while current:
                tag_name = current.name
                # 停止条件
                text = current.get_text(strip=True)
                if tag_name in ('p', 'div'):
                    if '上一条' in text:
                        break
                    if '返回列表' in text:
                        break
                if tag_name == 'a' and '返回列表' in text:
                    break
                if text and '发布时间' not in text:
                    content_parts.append(text)
                current = current.find_next_sibling()
        else:
            content_parts.append(soup.get_text(separator='\n', strip=True))

        content = "\n".join(content_parts)
        content = re.sub(r'\n{3,}', '\n\n', content).strip()
        content = f"<div class=\"cnjf-content\">\n{content}\n</div>"

        summary = content[:500] if content else title[:300]

        # 附件: 百度网盘
        attachments = []
        for a_tag in soup.find_all('a', href=re.compile(r'pan\.baidu\.com')):
            attachments.append({
                "name": a_tag.get_text(strip=True) or "百度网盘链接",
                "url": a_tag.get('href', '')
            })

        return title, content, pub_date, attachments, summary


if __name__ == '__main__':
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=4, help='页数')
    args = parser.parse_args()
    c = CnjfHjgsCrawler(
        site_name=SITE_NAME,
        domain=DOMAIN,
        url=BASE_URL,
        sync_delay=0.3,
    )
    c.max_pages = args.pages
    c.run()
