#!/usr/bin/env python3
import requests, re, time, os, sqlite3
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed
from playwright.sync_api import sync_playwright

BASE = "https://www.zh.gov.cn"
COL_URL = "https://www.zh.gov.cn/col/col1229821392/index.html"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "镇海区-生态环境分局-通知公告"
MAX_WORKERS = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}


def extract_content(html):
    """深度计数法提取 bt-content zoom clearfix 内部HTML"""
    m = re.search(r'<div class="bt-content zoom clearfix">', html)
    if not m:
        return ""
    start = m.end()
    depth = 1
    i = start
    while i < len(html) - 5 and depth > 0:
        if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r', '/'):
            # Only count opening divs, not self-closing
            if html[i+4] != '/':
                depth += 1
        elif html[i:i+6] == '</div>':
            depth -= 1
        i += 1
    return html[start:i-6].strip()


def process_detail(url):
    for att in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200 or not r.text:
                time.sleep(2)
                continue
            html = r.text

            # 标题: 优先 bt-art-title
            title = ""
            m = re.search(r'<p class="bt-art-title">(.*?)</p>', html)
            if m:
                title = m.group(1).strip()
            if not title:
                m = re.search(r'<title>(.*?)</title>', html)
                if m:
                    title = m.group(1).strip()

            # 正文: bt-content zoom clearfix 深度计数
            content = extract_content(html)
            if not content:
                # 备选: 尝试其他容器
                m = re.search(r'<div class="TRS_Editor"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
                if m:
                    content = m.group(1).strip()

            # 日期
            pub_date = ""
            m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
            if m:
                pub_date = m.group(1)

            full_url = url if url.startswith("http") else BASE + url
            return (full_url, title, content, pub_date)
        except Exception as e:
            if att < 2:
                time.sleep(2)
    return (url, None, None, None)


def collect_all_urls():
    all_urls = set()
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox", "--disable-gpu"])
        page = browser.new_page()
        page.goto(COL_URL, wait_until="commit", timeout=60000)
        time.sleep(3)
        for pg in range(1, 100):
            html = page.content()
            urls = re.findall(r'href="(/col/col1229821392/art/[^"]+\.html)"', html)
            for u in urls:
                all_urls.add(BASE + u)
            next_btn = page.query_selector("a.layui-laypage-next")
            if not next_btn:
                break
            dp = next_btn.get_attribute("data-page")
            if not dp or int(dp) <= pg:
                break
            next_btn.click()
            time.sleep(1.5)
        browser.close()
    return list(all_urls)


def main():
    mode = os.environ.get("MODE", "full")
    print(f"=== {SITE_NAME} (mode={mode}) ===", flush=True)

    if mode == "full":
        # 全量模式: 新收集URL，删旧数据重新跑
        print("Playwright收集URL...", flush=True)
        urls = collect_all_urls()
        print(f"  总URL: {len(urls)}条", flush=True)
        if not urls:
            print("  无数据", flush=True)
            return

        conn = sqlite3.connect(DB_PATH, timeout=60)
        cursor = conn.cursor()
        # 删除旧数据（内容为空）

        total_new = 0
        done = 0
        with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
            fut_map = {executor.submit(process_detail, url): url for url in urls}
            for fut in as_completed(fut_map):
                url, title, content, pub_date = fut.result()
                done += 1
                if done % 50 == 0:
                    print(f"  详情 {done}/{len(urls)}...", flush=True)
                if title is None:
                    continue
                if not content:
                    content = title
                    print(f"  [WARN] 无正文: {url}", flush=True)
                if not pub_date:
                    dm = re.search(r'/art/\d{4}/', url)
                    pub_date = f"{dm.group(0).split('/')[2]}-01-01" if dm else "2026-01-01"
                try:
                    d = date.fromisoformat(pub_date[:10])
                    if d < CUTOFF_DATE:
                        continue
                except:
                    pass
                try:
                    cursor.execute("""INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?)""",
                        (SITE_NAME, title, url, content, pub_date[:10], content, 0))
                    if cursor.rowcount > 0:
                        total_new += 1
                except Exception as e:
                    print(f"  [DB_ERROR] {e}", flush=True)
                if done % 50 == 0:
                    conn.commit()
        conn.commit()

        cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        cnt, min_d, max_d = cursor.fetchone()
        conn.close()
        print(f"\n=== 完成 ===", flush=True)
        print(f"  新增: {total_new}条", flush=True)
        print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)
    else:
        # 日跑模式: 只检查第1页
        print("日跑模式: 检查第1页", flush=True)
        # TODO: 简化，只取第1页增量


if __name__ == "__main__":
    main()
