#!/usr/bin/env python3
"""探测 v2: ahhz/ahtxq 试 label/8888 API, zixing 看列表结构"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. ahhz: 先 goto 栏目页拿 cookie, 再测 label/8888 API (两种 siteId)
        print("="*60)
        print("[ahhz] 栏目页 goto + API 测试")
        try:
            r = await page.goto("https://www.ahhz.gov.cn/zwgk/grassroots/column/6615882?&catId=1000260", timeout=45000, wait_until='domcontentloaded')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(2500)
            # 打印列表项 HTML 结构 (找带日期链接)
            items = await page.eval_on_selector_all(
                "a", "els => els.filter(a => /20\\d{2}[-年.]/.test(a.innerText||'') || /公告|公示|通知|批复|方案/.test(a.innerText||'')).slice(0,10).map(a => ({t:(a.innerText||'').trim().replace(/\\s+/g,' ').slice(0,60), h:a.href, cls:a.className.slice(0,40)}))"
            )
            for it in items:
                print(f"    item: [{it['t']}] cls={it['cls']} {it['h'][:90]}")
            # 试 API 两种 siteId
            for sid in ['6793374', '6793339']:
                api = (f"https://www.ahhz.gov.cn/huangshanzwgk/zwgk/site/label/8888?_=1"
                       f"&labelName=publicInfoList&siteId={sid}&organId=6615882&pageSize=15&pageIndex=1"
                       f"&isDate=true&dateFormat=yyyy-MM-dd&length=50&type=4&action=list&result="
                       f"&isJson=true&keyWords=&isSetValue=true&catIds=&catId=1000260")
                try:
                    rr = await page.evaluate(f"fetch('{api}').then(r=>r.text()).catch(e=>'FETCH_ERR:'+e.message)")
                    print(f"  API siteId={sid}: {str(rr)[:200]}")
                except Exception as e:
                    print(f"  API siteId={sid} EXC: {e}")
                await page.wait_for_timeout(1500)
        except Exception as e:
            print(f"  ERROR: {e}")

        # 2. ahtxq: 政务公开栏目页 goto
        print("="*60)
        print("[ahtxq] 政府信息公开栏目页")
        try:
            r = await page.goto("https://www.ahtxq.gov.cn/zwgk/public/column/6615872?type=2&nav=0", timeout=45000, wait_until='domcontentloaded')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(2500)
            items = await page.eval_on_selector_all(
                "a", "els => els.filter(a => /20\\d{2}[-年.]/.test(a.innerText||'')).slice(0,10).map(a => ({t:(a.innerText||'').trim().replace(/\\s+/g,' ').slice(0,60), h:a.href, cls:a.className.slice(0,40)}))"
            )
            for it in items:
                print(f"    item: [{it['t']}] cls={it['cls']} {it['h'][:90]}")
            # 试 API
            api = (f"https://www.ahtxq.gov.cn/huangshanzwgk/zwgk/site/label/8888?_=1"
                   f"&labelName=publicInfoList&siteId=6793339&organId=6615872&pageSize=15&pageIndex=1"
                   f"&isDate=true&dateFormat=yyyy-MM-dd&length=50&type=4&action=list&result="
                   f"&isJson=true&keyWords=&isSetValue=true&catIds=&catId=1000260")
            try:
                rr = await page.evaluate(f"fetch('{api}').then(r=>r.text()).catch(e=>'FETCH_ERR:'+e.message)")
                print(f"  API 试: {str(rr)[:200]}")
            except Exception as e:
                print(f"  API EXC: {e}")
        except Exception as e:
            print(f"  ERROR: {e}")

        # 3. zixing: 列表结构
        print("="*60)
        print("[zixing] 列表结构")
        try:
            r = await page.goto("http://www.zixing.gov.cn/zwgk/ztbd/zwgkgzydzl/zlhms/hjbf/hjzl/default.htm", timeout=45000, wait_until='domcontentloaded')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(2500)
            items = await page.eval_on_selector_all(
                "a", "els => els.filter(a => /20\\d{2}[-年.]/.test(a.innerText||'')).slice(0,12).map(a => ({t:(a.innerText||'').trim().replace(/\\s+/g,' ').slice(0,60), h:a.href, cls:a.className.slice(0,40)}))"
            )
            for it in items:
                print(f"    item: [{it['t']}] cls={it['cls']} {it['h'][:90]}")
            # 找分页
            pager = await page.eval_on_selector_all(
                "a", "els => els.filter(a => /下一页|末页|\\d+/.test((a.innerText||'').trim())).slice(0,10).map(a => ({t:(a.innerText||'').trim().slice(0,20), h:a.href}))"
            )
            for pg in pager:
                print(f"    pager: [{pg['t']}] {pg['h'][:90]}")
            # 找 total 文本
            totals = await page.evaluate("document.body.innerText.match(/共\\s*\\d+\\s*条|\\d+\\s*条/g)")
            print(f"  total 文本: {totals[:5] if totals else 'None'}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await browser.close()

asyncio.run(main())
