#!/usr/bin/env python3
import os
"""
郑州市生态环境局 - 政务公开 详情提取 (crawl_zz_hb.py)
CMS: 自建系统 + WAF (需Playwright)
分页: 507页, 10条/页, 总计~5070条
已通过 zz_collect.py 收集到 3689条 唯一URL于 /tmp/zz_articles.json
"""
import json, os, sys, time, sqlite3, re
from playwright.sync_api import sync_playwright

SITE_NAME = "zz_hb"
BASE_URL = "https://public.zhengzhou.gov.cn"
TEMP_DB = "/root/temp_search.db"

def fetch_detail(url, browser):
    """Fetch detail page content using Playwright"""
    context = browser.new_context(
        user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
    )
    page = context.new_page()
    try:
        page.goto(url, wait_until="domcontentloaded", timeout=30000)
        time.sleep(0.5)

        # Try multiple content selectors
        content_selectors = [
            "#zoom",
            ".bt_content",
            ".article_content",
            ".TRS_Editor",
            ".Custom_UnionStyle",
            ".content",
            "#UCAP-CONTENT",
            ".news-content",
            ".text-content",
            "div[class*=\"content\"]",
            "div[class*=\"article\"]",
            ".main-content",
            ".detail-content",
            "div[class*=\"detail\"]",
        ]

        content = ""
        for selector in content_selectors:
            try:
                el = page.query_selector(selector)
                if el:
                    content = el.inner_html()
                    text = re.sub(r'<[^>]+>', '', content).strip()
                    if len(text) >= 50:
                        break
                    content = ""
            except:
                continue

        # Fallback: body text
        if not content or len(re.sub(r'<[^>]+>', '', content).strip()) < 50:
            try:
                body = page.query_selector("body")
                if body:
                    all_html = body.inner_html()
                    all_html = re.sub(r'<header[^>]*>.*?</header>', '', all_html, flags=re.DOTALL)
                    all_html = re.sub(r'<footer[^>]*>.*?</footer>', '', all_html, flags=re.DOTALL)
                    all_html = re.sub(r'<nav[^>]*>.*?</nav>', '', all_html, flags=re.DOTALL)
                    text = re.sub(r'<[^>]+>', '', all_html).strip()
                    if len(text) >= 50:
                        content = f"<div>{all_html}</div>"
                    else:
                        text = page.inner_text("body") or ""
                        if len(text) >= 50:
                            content = f"<p>{text}</p>"
                        else:
                            return ""
            except:
                return ""
        else:
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)

        text_len = len(re.sub(r'<[^>]+>', '', content).strip())
        if text_len < 50:
            return ""

        return content

    except Exception as e:
        print(f"  ERROR: {e}")
        return ""
    finally:
        page.close()
        context.close()


def main():
    os.chdir("/root")

    with open("/tmp/zz_articles.json") as f:
        all_articles = json.load(f)

    total = len(all_articles)
    print(f"=== 郑州政务公开 详情提取 === 共{total}条")

    start_from = 0
    if os.path.exists("/tmp/zz_progress.txt"):
        with open("/tmp/zz_progress.txt") as f:
            start_from = int(f.read().strip())
        print(f"从第{start_from}条继续")

    conn = sqlite3.connect(TEMP_DB)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, pub_date TEXT
    )""")
    conn.commit()

    inserted = 0
    skipped = 0
    empty = 0
    last_progress = -1

    try:
        with sync_playwright() as p:
            browser = p.chromium.launch(headless=True)

            for i, art in enumerate(all_articles[start_from:], start_from):
                c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["url"],))
                if c.fetchone()[0] > 0:
                    skipped += 1
                    if i % 50 == 0:
                        print(f"  [{i}/{total}] inserted={inserted} skipped={skipped} empty={empty}")
                    continue

                content = fetch_detail(art["url"], browser)

                if not content:
                    empty += 1
                else:
                    try:
                        c.execute(
                            "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, pub_date) VALUES (?, ?, ?, ?, ?)",
                            (art["title"], content, art["url"], SITE_NAME, "")
                        )
                        conn.commit()
                        inserted += 1
                    except Exception as e:
                        print(f"  DB error: {e}")

                if i % 20 == 0 or i == total - 1:
                    print(f"  [{i}/{total}] inserted={inserted} skipped={skipped} empty={empty}")

                if i - last_progress >= 50 or i == total - 1:
                    with open("/tmp/zz_progress.txt", "w") as f:
                        f.write(str(i + 1))
                    print(f"  进度保存: {i+1}/{total}")
                    last_progress = i

                time.sleep(0.3)

    finally:
        if 'browser' in dir() and browser:
            browser.close()
        conn.close()

    print(f"\n=== 完成 ===")
    print(f"总计: {total}")
    print(f"新增: {inserted}")
    print(f"已存在: {skipped}")
    print(f"空正文(跳过): {empty}")


if __name__ == "__main__":
    main()
