#!/usr/bin/env python3
"""洪湖市-生态环境-通知公告 爬虫 (Playwright翻页 + requests详情)
"""
import json, re, time, os, sqlite3
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed
from playwright.sync_api import sync_playwright
import requests

BASE = "http://zwgk.honghu.gov.cn"
COL_URL = "http://zwgk.honghu.gov.cn/list.shtml?column_id=38559&page_index=1"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "洪湖市-生态环境-通知公告"
MAX_WORKERS = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}
# 总共8页，20条/页
TOTAL_PAGES = 8


def collect_all_urls_via_playwright():
    """使用Playwright翻页收集所有URL"""
    all_urls = []
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox", "--disable-gpu"])
        page = browser.new_page()
        page.goto(COL_URL, wait_until="domcontentloaded", timeout=30000)
        time.sleep(4)
        
        for pg in range(1, TOTAL_PAGES + 1):
            # 读取当前页列表
            items = page.query_selector_all('#documentsUL li a')
            for a in items:
                href = a.get_attribute('href')
                if href and 'honghu.gov.cn' in href:
                    all_urls.append(href)
                elif href and not href.startswith('http'):
                    full_url = BASE + href
                    if 'honghu.gov.cn' in full_url:
                        all_urls.append(full_url)
            
            print(f"  第{pg}页: {len(all_urls)} 条总URL", flush=True)
            
            if pg < TOTAL_PAGES:
                next_btn = page.query_selector(f'a[curpage="{pg+1}"]')
                if not next_btn:
                    next_btn = page.query_selector('a[curpage="2"]')  # 下一页按钮
                if next_btn:
                    next_btn.click()
                    time.sleep(2)
                else:
                    break
        
        browser.close()
    return all_urls


def extract_content_by_depth(html, class_name):
    """深度计数法提取指定class的div内部HTML"""
    # 支持多种class组合匹配
    pattern = r'<div[^>]*class="[^"]*' + re.escape(class_name) + r'[^"]*"[^>]*>'
    m = re.search(pattern, html)
    if not m:
        return ""
    start = m.end()
    depth = 1
    i = start
    while i < len(html) - 5 and depth > 0:
        if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r'):
            depth += 1
        elif html[i:i+6] == '</div>':
            depth -= 1
        i += 1
    return html[start:i-6].strip()


def process_detail(url):
    for att in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200 or not r.text:
                time.sleep(2)
                continue
            html = r.text
            
            # 标题: 优先 h1.zw-box-title
            title = ""
            m = re.search(r'<h1 class="zw-box-title">(.*?)</h1>', html)
            if m:
                title = m.group(1).strip()
            if not title:
                m = re.search(r'<title>(.*?)</title>', html)
                if m:
                    title = m.group(1).strip()
            # 去掉站点名后缀
            for suffix in ['-洪湖市人民政府', '_洪湖市人民政府', '--政府信息公开']:
                if title.endswith(suffix):
                    title = title[:-len(suffix)].strip()
            
            # 正文 - zw-center-txt
            content = extract_content_by_depth(html, "zw-center-txt")
            # 去掉开头的空预览占位（视频/音频/pdf）
            if content:
                content = re.sub(r'<!--.*?-->', '', content)
                content = re.sub(r'<div[^>]*>\s*</div>\s*', '', content)
                content = content.strip()
            if not content:
                content = extract_content_by_depth(html, "bt-content")
            if not content:
                content = extract_content_by_depth(html, "content")
            if not content:
                m = re.search(r'<div class="TRS_Editor"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
                if m:
                    content = m.group(1).strip()
            
            # 日期: 优先 zw-box-infos 中的发布时间
            pub_date = ""
            m = re.search(r'zw-box-infos[^>]*>.*?(\d{4}-\d{2}-\d{2})', html)
            if m:
                pub_date = m.group(1)
            if not pub_date:
                # 找 <i>发布时间</i> 后面的日期
                m = re.search(r'发布时间[：:]?\s*(\d{4}-\d{2}-\d{2})', html)
                if m:
                    pub_date = m.group(1)
            if not pub_date:
                m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
                if m:
                    pub_date = m.group(1)
            if not pub_date:
                # URL中提取: t116220263064 → 2026-06-16
                m = re.search(r'/t(\d{2})(\d{2})(\d{2})\d*/', url)
                if m:
                    pub_date = f"20{m.group(1)}-{m.group(2)}-{m.group(3)}"
            
            return (url, title, content, pub_date)
        except Exception as e:
            if att < 2:
                time.sleep(2)
    return (url, None, None, None)


def main():
    mode = os.environ.get("MODE", "full")
    print(f"=== {SITE_NAME} (mode={mode}) ===", flush=True)
    
    print("Playwright收集URL...", flush=True)
    urls = collect_all_urls_via_playwright()
    print(f"  总URL: {len(urls)}条", flush=True)
    if not urls:
        print("  无数据", flush=True)
        return
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cursor = conn.cursor()
    total_new = 0
    done = 0
    
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {executor.submit(process_detail, url): url for url in urls}
        for fut in as_completed(fut_map):
            url, title, content, pub_date = fut.result()
            done += 1
            if done % 20 == 0:
                print(f"  详情 {done}/{len(urls)}...", flush=True)
            if title is None:
                continue
            if not content:
                content = title
                print(f"  [WARN] 无正文: {title[:40]}", flush=True)
            if not pub_date:
                pub_date = "2026-01-01"
            try:
                d = date.fromisoformat(pub_date[:10])
                if d < CUTOFF_DATE:
                    continue
            except:
                pass
            try:
                cursor.execute("""INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, title, url, content, pub_date[:10], content, 0))
                if cursor.rowcount > 0:
                    total_new += 1
            except Exception as e:
                print(f"  [DB_ERROR] {e}", flush=True)
            if done % 50 == 0:
                conn.commit()
    conn.commit()
    
    cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()
    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
