#!/usr/bin/env python3
"""探测 v5: ahhz 完整字段+详情URL, ahtxq 找有内容栏目"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. ahhz: 完整 API 字段
        print("="*60)
        print("[ahhz] 完整字段")
        try:
            await page.goto("https://www.ahhz.gov.cn/zwgk/grassroots/column/6615882?&catId=1000260", timeout=45000, wait_until='domcontentloaded')
            await page.wait_for_timeout(3000)
            api = ("https://www.ahhz.gov.cn/huangshanzwgk/zwgk/site/label/8888?_=0.12345"
                   "&action=list&labelName=grassrootsContentPageList&siteId=6793339"
                   "&isDate=true&dateFormat=yyyy-MM-dd&length=50&organId=6615882&catId=1000260"
                   "&pageSize=15&pageIndex=1&type=4&keyWords=&isSetValue=true&isJson=true&result=")
            rr = await page.evaluate(f"fetch('{api}').then(r=>r.text()).catch(e=>'ERR:'+e.message)")
            data = json.loads(rr) if isinstance(rr, str) and rr.startswith('{') else None
            if data:
                print(f"  total: {data.get('total')} pageCount: {data.get('pageCount')}")
                first = data['data'][0]
                print(f"  字段: {sorted(first.keys())}")
                for k in ['title','link','contentId','publishDate','createDate','fileNum','author','attachRealName','attachSavedName']:
                    if k in first:
                        print(f"    {k} = {first[k]}")
        except Exception as e:
            print(f"  ERROR: {e}")

        # 2. ahhz: 详情 URL 测试 (从 data 拿 link/contentId)
        print("="*60)
        print("[ahhz] 详情 URL 测试")
        try:
            if data and data['data']:
                it = data['data'][0]
                cid = it.get('contentId') or it.get('id')
                link = it.get('link', '')
                print(f"  contentId={cid} link={link}")
                candidates = [
                    f"https://www.ahhz.gov.cn/zwgk/grassroots/public/{6615882}/{cid}.html",
                    f"https://www.ahhz.gov.cn/zwgk/public/{6615882}/{cid}.html",
                    f"https://www.ahhz.gov.cn/{cid}.html",
                    link if link.startswith('http') else f"https://www.ahhz.gov.cn{link}",
                ]
                for c in dict.fromkeys(candidates):
                    if not c: continue
                    try:
                        r = await page.goto(c, timeout=30000, wait_until='domcontentloaded')
                        await page.wait_for_timeout(1500)
                        t = await page.title()
                        wzcon = await page.evaluate("() => { const el = document.querySelector('.wzcon, .j-fontContent, .article-content, .content'); return el ? el.innerText.slice(0,80).replace(/\\s+/g,' ') : null }")
                        print(f"  [{c[:80]}] -> {r.status if r else '?'} title={t[:40]} body={wzcon[:60] if wzcon else 'NONE'}")
                    except Exception as e:
                        print(f"  [{c[:80]}] EXC {str(e)[:60]}")
                    await page.wait_for_timeout(800)
        except Exception as e:
            print(f"  ERROR: {e}")

        await page.wait_for_timeout(1500)

        # 3. ahtxq: 找有内容的栏目 (看页面栏目树)
        print("="*60)
        print("[ahtxq] 找有内容栏目")
        try:
            await page.goto("https://www.ahtxq.gov.cn/zwgk/public/column/6615872?type=2&nav=0", timeout=45000, wait_until='domcontentloaded')
            await page.wait_for_timeout(3000)
            # 抓包到的 catIds 参数
            cats = await page.evaluate("""() => {
                const out = [];
                document.querySelectorAll('a').forEach(a => {
                    const h = a.href;
                    const t = (a.innerText||'').trim().slice(0,20);
                    if (h.includes('column/') && t) out.push({t, h});
                });
                return out.slice(0,30);
            }""")
            seen = set()
            for c in cats:
                if c['h'] not in seen and ('生态环境' in c['t'] or '环境' in c['t'] or '水利' in c['t'] or '住建' in c['t'] or '审批' in c['t']):
                    seen.add(c['h'])
                    print(f"  col: [{c['t']}] {c['h'][:100]}")
            if not seen:
                for c in cats[:15]:
                    print(f"  col: [{c['t']}] {c['h'][:100]}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await browser.close()

asyncio.run(main())
