#!/usr/bin/env python3
"""探测 ahtxq: 找生态环境分局栏目 organId/catId"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. ahtxq 首页找生态环境/部门链接
        print("="*60)
        print("[ahtxq] 首页找部门/环保链接")
        await page.goto("https://www.ahtxq.gov.cn/index.html", timeout=45000, wait_until='domcontentloaded')
        await page.wait_for_timeout(6000)
        links = await page.evaluate("""() => {
            const out = [];
            document.querySelectorAll('a').forEach(a => {
                const t = (a.innerText||'').trim().replace(/\\s+/g,'');
                if (t && t.length<20 && (t.includes('生态环境')||t.includes('环保')||t.includes('部门')||t.includes('公开')||t.includes('镇')||t.includes('街道'))) {
                    out.push({t, h: a.href});
                }
            });
            return out.slice(0,30);
        }""")
        seen=set()
        for l in links:
            if l['h'] not in seen:
                seen.add(l['h'])
                print(f"  link: [{l['t']}] {l['h'][:110]}")

        # 2. 直接试环保分局栏目 URL (屯溪区 organId 可能 6615xxx 段)
        print("="*60)
        print("[ahtxq] 试生态环境分局 organId")
        for organ in ['6616285','6616268','6616370','6616368','6616369','6615872']:
            url = f"https://www.ahtxq.gov.cn/zwgk/public/column/{organ}?type=4&catId=6719037&action=list&nav=3"
            try:
                r = await page.goto(url, timeout=20000, wait_until='domcontentloaded')
                await page.wait_for_timeout(1500)
                t = await page.title()
                # 检查是否有 API 调用 (publicInfoList)
                api_hit = []
                async def cb(req):
                    if 'publicInfoList' in req.url:
                        api_hit.append(req.url)
                page.on('request', cb)
                await page.wait_for_timeout(2000)
                page.remove_listener('request', cb)
                print(f"  [{organ}] {r.status if r else '?'} title={t[:40]} api={len(api_hit)}")
                if api_hit:
                    print(f"    API: {api_hit[0][:160]}")
            except Exception as e:
                print(f"  [{organ}] EXC {str(e)[:60]}")
            await page.wait_for_timeout(800)

        # 3. 搜"屯溪区生态环境分局"站内
        print("="*60)
        print("[ahtxq] 站内搜索")
        try:
            search_url = "https://www.ahtxq.gov.cn/site/search/6793339?isAllSite=true&siteId=6793339&searchWord=生态环境分局"
            r = await page.goto(search_url, timeout=25000, wait_until='domcontentloaded')
            await page.wait_for_timeout(4000)
            res = await page.evaluate("""() => {
                const out = [];
                document.querySelectorAll('a').forEach(a => {
                    const t = (a.innerText||'').trim();
                    if (t && t.includes('生态环境') && t.length<40) out.push({t: t.slice(0,40), h: a.href});
                });
                return out.slice(0,10);
            }""")
            for r_ in res:
                print(f"  hit: [{r_['t']}] {r_['h'][:110]}")
        except Exception as e:
            print(f"  EXC: {str(e)[:80]}")

        await browser.close()

asyncio.run(main())
