#!/usr/bin/env python3
"""
池州市贵池区 - 环评公示 (ahgc.gov.cn)
列表: /Jczwgk/opennessList/257/111001001/page_N.html
详情: /Jczwgk/opennessShow/N.html
正文: div.m-dttexts#zoom (inside div.m-detailbox)
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "贵池区-环评公示"
BASE_URL = "https://www.ahgc.gov.cn"
LIST_URL = "https://www.ahgc.gov.cn/Jczwgk/opennessList/257/111001001/page_{n}.html"
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except:
        return None


def parse_list(html):
    """
    从列表页 HTML 提取 (title, url, date) 元组
    结构: <ul><li><span>2026-06-01</span><a href="..." title="...">title</a></li></ul>
    """
    from bs4 import BeautifulSoup
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    # 找到包含新闻/公告列表的 ul
    for ul in soup.find_all('ul'):
        lis = ul.find_all('li')
        if len(lis) < 3:
            continue
        # 检查是否包含 span+date+a 结构
        candidate = []
        for li in lis:
            span = li.find('span')
            a = li.find('a', href=True)
            if span and a and span.get_text(strip=True) and a.get('href', ''):
                date_text = span.get_text(strip=True)
                # 检查日期格式
                if re.match(r'\d{4}-\d{1,2}-\d{1,2}', date_text):
                    candidate.append((a, date_text))
        if len(candidate) >= 3:
            # 这是目标列表
            for a, date_text in candidate:
                href = a.get('href', '').strip()
                title = a.get('title', '') or a.get_text(strip=True)
                if not href or not title or len(title) < 5:
                    continue
                if not href.startswith('http'):
                    href = BASE_URL + href
                items.append((title, href, date_text))
            break  # 找到即退出

    return items


def fetch_detail(url):
    """访问详情页，返回 (title, content, pub_date)"""
    html = fetch(url)
    if not html:
        return None, None, None

    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    result = {}

    # 标题：优先 h1.u-lgtit，其次 <title>
    h1 = soup.find('h1', class_='u-lgtit')
    if h1:
        result['title'] = h1.get_text(strip=True)
    if not result.get('title'):
        tag = soup.find('title')
        if tag:
            t = tag.get_text(strip=True)
            # 去掉站点后缀
            t = re.sub(r'[-—].*?(?:池州市|贵池区).*?(?:人民政府|政府).*$', '', t).strip()
            result['title'] = t

    # 日期：meta PubDate 优先
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', meta['content'])
        if m:
            result['publish_date'] = m.group(1)
    # 降级：发布时间：
    if not result.get('publish_date'):
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m:
            result['publish_date'] = m.group(1)

    # 正文：div.m-dttexts#zoom (或降级到 div.m-detailbox 内文本)
    detail = soup.find('div', class_='m-dttexts')
    if not detail:
        detail = soup.find('div', class_='j-fontContent')
    if not detail:
        detail = soup.find('div', id='zoom')
    if not detail:
        detail = soup.find('div', class_='m-detailbox')
    if not detail:
        # 兜底：class="content"
        detail = soup.find('div', class_='content')

    if detail:
        content_html = str(detail)
        # 清理 script/style
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL | re.I)
        content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL | re.I)
        result['content'] = content_html.strip()

    return result.get('title'), result.get('content'), result.get('publish_date')


def main():
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")

    # 爬列表页
    all_list = []
    for page in range(1, MAX_PAGES + 1):
        url = LIST_URL.replace('{n}', str(page))
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("❌ 请求失败")
            break
        items = parse_list(html)
        if not items:
            print("0 条（可能已到底）")
            break
        print(f"✅ {len(items)} 条")
        all_list.extend(items)

    print(f"\n📊 列表总计: {len(all_list)} 条")

    # 爬详情
    all_items, seen = [], set()
    for i, (title, url, list_date) in enumerate(all_list):
        if url in seen:
            continue
        seen.add(url)

        print(f"  [{i+1}/{len(all_list)}] {title[:50]}...", end=" ", flush=True)

        dt, content, date = fetch_detail(url)
        final_title = dt or title
        final_date = date or list_date

        # 统一日期格式 YYYY-MM-DD
        if final_date:
            m = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', str(final_date))
            if m:
                y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
                final_date = f"{y:04d}-{mo:02d}-{d:02d}"
            else:
                final_date = ""

        # 纯文本摘要
        content_text = content or ""
        summary = re.sub(r'<[^>]+>', ' ', content_text).strip()
        summary = re.sub(r'\s+', ' ', summary)[:300]

        all_items.append({
            "site_name": SITE_NAME,
            "title": final_title,
            "url": url,
            "content": content_text,
            "pub_date": final_date,
            "summary": summary,
            "tags": SITE_NAME,
        })
        print("✅")
        time.sleep(0.3)

    # 推送
    if all_items:
        push_to_searchdb(all_items, "ahgc_hpgs")
    else:
        print("  ⏭ 无数据，跳过推送")

    print(f"\n✅ 完成! 共 {len(all_items)} 条")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
