#!/usr/bin/env python3
"""探测 ahtxq: 信息公开目录找生态环境栏目"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. 信息公开目录 type=4
        print("="*60)
        print("[ahtxq] 信息公开目录")
        await page.goto("https://www.ahtxq.gov.cn/zwgk/public/column/6615872?type=4&action=list", timeout=45000, wait_until='domcontentloaded')
        await page.wait_for_timeout(6000)
        # 抓所有 column 链接 + API 请求
        api_hits = []
        page.on('request', lambda req: api_hits.append(req.url) if 'publicInfoList' in req.url or 'label/8888' in req.url else None)
        await page.wait_for_timeout(4000)
        print(f"  API 请求 ({len(api_hits)}):")
        for u in api_hits[:10]:
            print(f"    {u[:180]}")
        # 栏目树
        cols = await page.evaluate("""() => {
            const out = [];
            document.querySelectorAll('a').forEach(a => {
                const t = (a.innerText||'').trim().replace(/\\s+/g,'');
                const h = a.href;
                if (t && t.length<25 && h.includes('/zwgk/') && (h.includes('column') || h.includes('col/'))) {
                    out.push({t, h});
                }
            });
            return out;
        }""")
        seen = set()
        for c in cols:
            if c['h'] not in seen:
                seen.add(c['h'])
                if any(k in c['t'] for k in ['环境','生态','水利','住建','审批','应急','规划']):
                    print(f"  col: [{c['t']}] {c['h'][:120]}")
        if not seen:
            print("  (无栏目链接)")

        # 2. 抓取列表页所有内容链接 (tzgg 通知公告)
        print("="*60)
        print("[ahtxq] tzgg 通知公告样本")
        items = await page.evaluate("""() => {
            const out = [];
            document.querySelectorAll('a').forEach(a => {
                const t = (a.innerText||'').trim();
                const h = a.href;
                if (t && /20\\d{2}/.test(t) && (t.includes('公示')||t.includes('公告')||t.includes('通知')||t.includes('批复')||t.includes('方案')) && h.includes('ahtxq') && !h.includes('column')) {
                    out.push({t: t.slice(0,50), h});
                }
            });
            return out.slice(0,15);
        }""")
        for it in items:
            print(f"  item: [{it['t']}] {it['h'][:100]}")

        # 3. 详情页结构 (9327358)
        print("="*60)
        print("[ahtxq] 详情页结构")
        try:
            r = await page.goto("https://www.ahtxq.gov.cn/zxzx/tzgg/9327358.html", timeout=30000, wait_until='domcontentloaded')
            await page.wait_for_timeout(3000)
            print(f"  status={r.status if r else '?'}")
            print(f"  title: {(await page.title())[:70]}")
            metas = await page.eval_on_selector_all("meta", "els => els.filter(m => /ArticleTitle|PubDate|ContentSource/i.test(m.name||'')).map(m => ({n:m.name, c:(m.content||'').slice(0,50)}))")
            for m in metas:
                print(f"    meta: {m}")
            for sel in ['.wzcon','.j-fontContent','.article-content','.content','#content','.TRS_Editor']:
                cnt = await page.evaluate(f"() => {{ const el = document.querySelector('{sel}'); return el ? el.innerText.slice(0,80).replace(/\\s+/g,' ') : null }}")
                if cnt:
                    print(f"    [{sel}] {cnt[:70]}")
            tbls = await page.evaluate("document.querySelectorAll('table').length")
            print(f"  tables: {tbls}")
        except Exception as e:
            print(f"  EXC: {str(e)[:80]}")

        await browser.close()

asyncio.run(main())
