#!/usr/bin/env python3
"""探测 ahtxq: 面包屑栏目 + 通知公告列表"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. 详情页面包屑
        print("="*60)
        print("[ahtxq] 详情页面包屑/栏目")
        await page.goto("https://www.ahtxq.gov.cn/zxzx/tzgg/9327358.html", timeout=30000, wait_until='domcontentloaded')
        await page.wait_for_timeout(3000)
        crumbs = await page.evaluate("""() => {
            const out = [];
            document.querySelectorAll('.crumbs a, .breadcrumb a, .location a, .dqwz a, .nav-crumb a').forEach(a => {
                out.push({t: (a.innerText||'').trim().slice(0,25), h: a.href});
            });
            return out;
        }""")
        for c in crumbs:
            print(f"  crumb: [{c['t']}] {c['h'][:110]}")
        # 找页面里所有栏目链接
        col_links = await page.evaluate("""() => {
            const out = [];
            document.querySelectorAll('a').forEach(a => {
                const t = (a.innerText||'').trim();
                if (t && t.length<20 && a.href.includes('/zwgk/public/column/')) out.push({t, h: a.href});
            });
            return out.slice(0,15);
        }""")
        for c in col_links:
            print(f"  col: [{c['t']}] {c['h'][:110]}")

        # 2. 通知公告列表页
        print("="*60)
        print("[ahtxq] tzgg 通知公告列表")
        for url in ["https://www.ahtxq.gov.cn/zxzx/tzgg/index.html", "https://www.ahtxq.gov.cn/zxzx/tzgg/"]:
            try:
                r = await page.goto(url, timeout=30000, wait_until='domcontentloaded')
                await page.wait_for_timeout(3000)
                print(f"  [{url}] {r.status if r else '?'}")
                items = await page.evaluate("""() => {
                    const out = [];
                    document.querySelectorAll('a').forEach(a => {
                        const t = (a.innerText||'').trim();
                        if (t && t.length>8 && a.href.includes('ahtxq') && /\\d{6,}\\.html/.test(a.href)) {
                            out.push({t: t.slice(0,45), h: a.href});
                        }
                    });
                    return out.slice(0,10);
                }""")
                for it in items:
                    print(f"    item: [{it['t']}] {it['h'][:90]}")
                if items: break
            except Exception as e:
                print(f"  EXC: {str(e)[:60]}")

        # 3. 站内搜索 API (黄山体系)
        print("="*60)
        print("[ahtxq] 站内搜索生态环境")
        try:
            api = "https://www.ahtxq.gov.cn/site/search/6793339?isAllSite=true&siteId=6793339&searchWord=生态环境&pageSize=10"
            r = await page.goto(api, timeout=30000, wait_until='domcontentloaded')
            await page.wait_for_timeout(3500)
            print(f"  status={r.status if r else '?'} title={(await page.title())[:40]}")
            hits = await page.evaluate("""() => {
                const out = [];
                document.querySelectorAll('a').forEach(a => {
                    const t = (a.innerText||'').trim();
                    if (t && t.includes('生态环境') && t.length<45) out.push({t: t.slice(0,45), h: a.href});
                });
                return out.slice(0,12);
            }""")
            for h_ in hits:
                print(f"  hit: [{h_['t']}] {h_['h'][:110]}")
            if not hits:
                body = await page.evaluate("document.body.innerText.slice(0,300).replace(/\\s+/g,' ')")
                print(f"  body: {body[:250]}")
        except Exception as e:
            print(f"  EXC: {str(e)[:80]}")

        await browser.close()

asyncio.run(main())
