#!/usr/bin/env python3
"""
xuanzhou.gov.cn 宣州区政府信息公开 - 环评公示爬虫
==================================================
列表页需 Playwright（WAF屏蔽），详情页可直接 requests 访问。
类别: 909（最新公开-宣城高新区管委会）

用法:
  python3 crawl_xuanzhou.py                    # 增量爬（第1页）
  python3 crawl_xuanzhou.py --full             # 全量（仅1页）
"""

import os, re, sys, time, json, sqlite3
import requests
from bs4 import BeautifulSoup

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "宣州区人民政府"
DOMAIN = "www.xuanzhou.gov.cn"
CATEGORY_ID = 909
CATEGORY_NAME = "最新公开-环评公示"
GROUP = "安徽-宣城"
INDUSTRY = "环评公示"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

LIST_URL = "https://www.xuanzhou.gov.cn/XxgkContent/showList/%d/0/page_1.html" % CATEGORY_ID
DETAIL_BASE = "https://www.xuanzhou.gov.cn"
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch_list_via_playwright():
    """Playwright获取列表页内容（绕过WAF）"""
    from playwright.sync_api import sync_playwright
    
    items = []
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox"])
        ctx = browser.new_context(
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
            locale="zh-CN",
        )
        page = ctx.new_page()
        try:
            page.goto(LIST_URL, timeout=60000, wait_until="networkidle")
            page.wait_for_timeout(3000)
            html = page.content()
            soup = BeautifulSoup(html, "lxml")
            
            # 提取所有内容链接
            for a in soup.find_all("a"):
                href = a.get("href", "")
                if "/OpennessContent/show/" in href:
                    title = a.get_text(strip=True)
                    full_url = href if href.startswith("http") else DETAIL_BASE + href
                    items.append((title, full_url))
            
            # 去重（保留顺序）
            seen = set()
            deduped = []
            for t, u in items:
                if u not in seen:
                    seen.add(u)
                    deduped.append((t, u))
            items = deduped
        except Exception as e:
            print("  Playwright error: %s" % e, file=sys.stderr)
        finally:
            browser.close()
    return items


def fetch_detail(url):
    """获取详情页内容"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        
        # 标题 - 优先从meta取
        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
        
        # 发布日期
        date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            date = meta_d["content"].strip()[:10]
        if not date:
            # 从表格取发布日期
            tb = soup.select_one("table.m-detailtb")
            if tb:
                for td in tb.find_all("td"):
                    txt = td.get_text(strip=True)
                    if "发布日期" in txt:
                        next_td = td.find_next_sibling("td")
                        if next_td:
                            date = next_td.get_text(strip=True)[:10]
        
        # 正文 - #zoom
        body_html = ""
        summary = ""
        zoom = soup.select_one("#zoom")
        if zoom:
            for tag in zoom.find_all(["script", "style"]):
                tag.decompose()
            body_html = str(zoom)
            summary = zoom.get_text(strip=True)[:200]
        
        return title, date, body_html, summary
    except Exception as e:
        print("    detail error %s: %s" % (url[-40:], e), file=sys.stderr)
        return "", "", "", ""


def store_record(title, page_url, publish_date, body_html="", summary=""):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title.strip(), page_url, DOMAIN, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, body_html, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0
        
        if not is_new and body_html:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (body_html, summary, row[0]))
                conn.commit()
                is_new = True
        
        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                _c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    _c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name) VALUES (?,?,?)",
                                (row[0], title.strip(), SITE_NAME))
                    _c2.commit()
                except:
                    pass
                finally:
                    _c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description="xuanzhou 环评公示爬虫")
    parser.add_argument("--full", action="store_true", help="全量爬")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数（仅1页）")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    print("📄 获取列表页（Playwright）...")
    items = fetch_list_via_playwright()
    print("🔍 列表共 %d 条" % len(items))

    for idx, (title, url) in enumerate(items):
        print("  [%d/%d] %s" % (idx + 1, len(items), title[:50]))
        dt_title, dt_date, body, summary = fetch_detail(url)
        
        # 优先用详情页的数据
        final_title = dt_title or title
        final_date = dt_date or ""
        
        store_record(final_title, url, final_date, body, summary)
        time.sleep(0.5)

    print("\n✅ 完成! 新%d, 跳过%d, 错误%d" % (stats["new"], stats["skip"], stats["errors"]))
