#!/usr/bin/env python3
"""
太和县 - 县经开区其他 (taihe.gov.cn)
列表: /OpennessTarget/423/36712/page_N.html  → <ul><li><a><span>日期</span></li>
详情: /OpennessContent/show/N.html
正文: div.m-detailbox
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "太和县-县经开区其他"
BASE_URL = "https://www.taihe.gov.cn"
LIST_API = "https://www.taihe.gov.cn/OpennessTarget/423/36712/page_{n}.html"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except:
        return None


def parse_list(html):
    """从列表页 HTML 的 <ul><li><a><span> 结构中提取 (title, url, date) 元组"""
    from bs4 import BeautifulSoup
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul')
    if not ul:
        return items
    lis = ul.find_all('li')
    for li in lis:
        a = li.find('a')
        span = li.find('span')
        if not a:
            continue
        href = a.get('href', '').strip()
        title = a.get('title', '') or a.get_text(strip=True)
        date = span.get_text(strip=True) if span else ""
        if not href or not title or len(title) < 5:
            continue
        if not href.startswith('http'):
            href = BASE_URL + href
        items.append((title, href, date))
    return items


def fetch_detail(url):
    """访问详情页，返回 (title, content, pub_date)"""
    html = fetch(url)
    if not html:
        return None, None, None

    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    result = {}

    # 标题
    title_div = soup.find('div', class_='u-title')
    if title_div:
        result['title'] = title_div.get_text(strip=True)

    # 日期：先用 meta
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', meta['content'])
        if m:
            result['publish_date'] = m.group(1)
    # 再用 发布时间：
    if not result.get('publish_date'):
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m:
            result['publish_date'] = m.group(1)

    # 正文
    detail = soup.find('div', class_='m-detailbox')
    if detail:
        content_html = str(detail)
        # 清理 script/style
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL|re.I)
        content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL|re.I)
        result['content'] = content_html.strip()

    return result.get('title'), result.get('content'), result.get('publish_date')


def main():
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")

    # 爬列表页
    all_list = []
    for page in range(1, MAX_PAGES + 1):
        url = LIST_API.replace('{n}', str(page))
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("❌ 请求失败")
            break
        items = parse_list(html)
        if not items:
            print("0 条（可能已到底）")
            break
        print(f"✅ {len(items)} 条")
        all_list.extend(items)

    print(f"\n📊 列表总计: {len(all_list)} 条")

    # 爬详情
    all_items, seen = [], set()
    for i, (title, url, list_date) in enumerate(all_list):
        if url in seen:
            continue
        seen.add(url)

        print(f"  [{i+1}/{len(all_list)}] {title[:50]}...", end=" ", flush=True)

        dt, content, date = fetch_detail(url)
        final_title = dt or title
        final_date = date or list_date

        # 清理正文 HTML → 纯文本摘要
        content_text = content or ""
        summary = re.sub(r'<[^>]+>', ' ', content_text).strip()
        summary = re.sub(r'\s+', ' ', summary)[:300]

        all_items.append({
            "site_name": SITE_NAME,
            "title": final_title,
            "url": url,
            "content": content_text,
            "pub_date": final_date,
            "summary": summary,
            "tags": SITE_NAME,
        })
        print("✅")
        time.sleep(0.3)

    # 推送
    if all_items:
        push_to_searchdb(all_items, "taihe_zxjd")
    else:
        print("  ⏭ 无数据，跳过推送")

    print(f"\n✅ 完成! 共 {len(all_items)} 条")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
