#!/usr/bin/env python3
"""探测 v7: ahhz 详情正文容器 + ahtxq 生态环境栏目定位"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. ahhz 详情页
        print("="*60)
        print("[ahhz] 详情页结构")
        try:
            r = await page.goto("https://www.ahhz.gov.cn/zwgk/grassroots/6615882/12150402.html", timeout=45000, wait_until='domcontentloaded')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(3000)
            print(f"  title: {(await page.title())[:80]}")
            metas = await page.eval_on_selector_all("meta", "els => els.filter(m => /ArticleTitle|PubDate|ContentSource/i.test(m.name||'')).map(m => ({n:m.name, c:(m.content||'').slice(0,60)}))")
            for m in metas:
                print(f"    meta: {m}")
            for sel in ['.wzcon', '.j-fontContent', '.article-content', '.content', '#content', '.TRS_Editor', '.zwcontent', '.xl_content']:
                cnt = await page.evaluate(f"() => {{ const el = document.querySelector('{sel}'); return el ? el.innerText.slice(0,120).replace(/\\s+/g,' ') : null }}")
                if cnt:
                    print(f"    [{sel}] {cnt[:100]}")
            tbls = await page.evaluate("document.querySelectorAll('table').length")
            print(f"  tables: {tbls}")
            # 附件链接
            atts = await page.eval_on_selector_all("a", "els => els.filter(a => /\\.(pdf|docx?|xlsx?|zip|rar)/i.test(a.href) && a.href.includes('ahhz')).map(a => ({t:(a.innerText||'').trim().slice(0,40), h:a.href.slice(0,100)})).slice(0,5)")
            for a in atts:
                print(f"    att: [{a['t']}] {a['h']}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await page.wait_for_timeout(1500)

        # 2. ahtxq: 搜生态环境分局栏目 (站内搜索/单位列表)
        print("="*60)
        print("[ahtxq] 找生态环境栏目")
        try:
            await page.goto("https://www.ahtxq.gov.cn/zwgk/public/column/6615872?type=2&nav=0", timeout=45000, wait_until='domcontentloaded')
            await page.wait_for_timeout(4000)
            # 尝试搜索
            html = await page.content()
            # 找 部门/单位 链接
            links = await page.evaluate("""() => {
                const out = [];
                document.querySelectorAll('a').forEach(a => {
                    const t = (a.innerText||'').trim().replace(/\\s+/g,'');
                    if (t && (t.includes('生态环境') || t.includes('环保') || t.includes('部门') || t.includes('单位') || t.includes('乡镇')) && a.href.includes('ahtxq')) {
                        out.push({t: t.slice(0,25), h: a.href});
                    }
                });
                return out.slice(0,20);
            }""")
            for l in links:
                print(f"  link: [{l['t']}] {l['h'][:110]}")
            if not links:
                # 站内搜索
                api = ("https://www.ahtxq.gov.cn/huangshanzwgk/zwgk/site/label/8888?_=0.55"
                       "&labelName=publicCatalogAssigned&siteId=6793339&organId=6615872&isJson=true&type=2")
                rr = await page.evaluate(f"fetch('{api}').then(r=>r.text()).catch(e=>'ERR:'+e.message)")
                print(f"  catalog API: {str(rr)[:400]}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await browser.close()

asyncio.run(main())
