#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
威海南海新区政务服务平台 - 通知公告 (whnhzwfw.sd.gov.cn/nanhai/govservice/notice)
站点: 山东政务服务网 威海市·南海新区 (icity 框架)
列表: /nanhai/govservice/notice -> JS onQuery(N) 翻页 (limit=8, 实际渲染10含置顶)
      API: POST /nanhai/api-v2/app.icity.govservice.GovProjectCmd/getPoliciList/execute
      (AES-ECB加密 + URL签名, 直接在页面上下文执行 onQuery 绕开)
详情: /nanhai/icity/publishdetail?id={uuid}
      标题 #title, 发布机关 #deptname, 成文日期 #writetime, 发布日期 #ctime, 正文 #contentDetail
注意: https 被 TLS 拦截, 必须用 http + playwright stealth
"""
import asyncio, json, re, sys, argparse
from playwright.async_api import async_playwright
from playwright_stealth import Stealth

BASE = "http://whnhzwfw.sd.gov.cn"
LIST_URL = f"{BASE}/nanhai/govservice/notice"
SITE_NAME = "威海南海新区-通知公告"

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb  # noqa

async def fetch_detail(page, url, sem):
    """抓详情页: 标题/日期/正文"""
    async with sem:
        for attempt in range(3):
            try:
                await asyncio.wait_for(page.goto(url, wait_until="domcontentloaded", timeout=45000), timeout=50)
                await page.wait_for_timeout(800)
                info = await page.evaluate('''() => {
                    const g = (id) => { const el = document.getElementById(id); return el ? el.textContent.trim() : ""; };
                    const body = document.getElementById("contentDetail");
                    let html = "";
                    if (body) {
                        // 移除脚本/style, 保留表格与段落
                        const clone = body.cloneNode(true);
                        clone.querySelectorAll("script,style,iframe,object").forEach(e => e.remove());
                        html = clone.innerHTML;
                    }
                    return { title: g("title"), dept: g("deptname"), ctime: g("ctime"), writetime: g("writetime"), html: html };
                }''')
                if info["html"] or info["title"]:
                    return info
            except Exception as e:
                if attempt == 2:
                    print(f"    [DETAIL ERR] {type(e).__name__} {str(e)[:120]}")
                else:
                    await page.wait_for_timeout(2000 * (attempt + 1))
        return None

async def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=15, help="最大页数")
    parser.add_argument("--max", type=int, default=0, help="最大条数(0=全部)")
    args = parser.parse_args()

    results = []
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=["--no-sandbox"])
        ctx = await browser.new_context(
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
            viewport={"width": 1366, "height": 900}, locale="zh-CN")
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 打开列表页(触发自动加载第1页) - TLS偶发拦截, 加重试
        for attempt in range(5):
            try:
                await asyncio.wait_for(page.goto(LIST_URL, wait_until="domcontentloaded", timeout=45000), timeout=50)
                await page.wait_for_timeout(4000)
                ok = await page.evaluate("typeof onQuery === 'function'")
                if ok:
                    break
                print(f"  [RETRY] page {attempt+1} loaded but no onQuery")
            except Exception as e:
                print(f"  [RETRY] {attempt+1}/5 {type(e).__name__} {str(e)[:100]}")
                if attempt == 4:
                    raise
                await page.wait_for_timeout(5000 * (attempt + 1))

        # 读取列表页并翻页
        seen = set()
        for pg in range(1, args.pages + 1):
            if pg > 1:
                await page.evaluate(f"onQuery({pg})")
                await page.wait_for_timeout(2500)
            rows = await page.eval_on_selector_all("#roll_list1 tr", '''els => els.map(e => {
                const a = e.querySelector("a");
                const tds = e.querySelectorAll("td");
                return { t: a ? a.textContent.trim() : "", h: a ? a.href : "", date: tds.length > 1 ? tds[tds.length-1].textContent.trim() : "" };
            })''')
            new_cnt = 0
            for r in rows:
                t = r["t"].strip()
                h = r["h"].strip()
                if not h or not t:
                    continue
                # 跳过置顶外链(威海市级)
                if "whzwfw.sd.gov.cn/wh/" in h:
                    continue
                if h in seen:
                    continue
                seen.add(h)
                results.append({"title": t, "url": h, "date": r["date"]})
                new_cnt += 1
            print(f"[LIST] page {pg}: rows={len(rows)} new={new_cnt} total={len(results)}")
            if args.max and len(results) >= args.max:
                break
            if new_cnt == 0 and pg > 1:
                break
            # 判断是否到末页
            total = await page.evaluate("layerpageAllTotal || 0")
            if total and len(seen) >= total:
                break

        print(f"共 {len(results)} 条列表, 开始抓详情...")
        # 详情页用新 page
        page2 = await ctx.new_page()
        sem = asyncio.Semaphore(2)
        done = 0
        for item in results:
            info = await fetch_detail(page2, item["url"], sem)
            done += 1
            if info:
                item["detail"] = info
                print(f"  [{done}/{len(results)}] {info['title'][:40]}")
            else:
                item["detail"] = None
                print(f"  [{done}/{len(results)}] FAIL {item['title'][:40]}")
        await browser.close()

    # 入库
    records = []
    for it in results:
        d = it.get("detail") or {}
        title = d.get("title") or it["title"]
        date = d.get("ctime") or it["date"] or ""
        content = d.get("html", "")
        if not content:
            content = f"<p>{it['title']}</p>"
        records.append({
            "site_name": SITE_NAME,
            "title": title,
            "source_url": it["url"],
            "url": it["url"],
            "content": content,
            "pub_date": date,
            "summary": title[:200],
            "source": d.get("dept", ""),
        })
    push_to_searchdb(records, batch_label="whnh_tzgg")
    print(f"Done: {len(records)} records")

if __name__ == "__main__":
    asyncio.run(main())
