#!/usr/bin/env python3
"""探测 v3: zixing 详情结构 + ahhz/ahtxq 抓包找 API"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. zixing 详情页
        print("="*60)
        print("[zixing] 详情页结构")
        try:
            r = await page.goto("http://www.zixing.gov.cn/zwgk/ztbd/zwgkgzydzl/zlhms/hjbf/hjzl/content_4028850.html", timeout=45000, wait_until='domcontentloaded')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(2500)
            title = await page.title()
            print(f"  title: {title[:80]}")
            # meta 信息
            metas = await page.eval_on_selector_all(
                "meta", "els => els.filter(m => /title|date|source|PubDate|Article/i.test(m.name||'')).map(m => ({n:m.name, c:(m.content||'').slice(0,60)}))"
            )
            for m in metas[:8]:
                print(f"    meta: {m}")
            # 正文容器候选
            for sel in ['.wzcon', '.j-fontContent', '.article-content', '.content', '#content', '.TRS_Editor', '.view', '.xl_content', 'div.zwxl']:
                cnt = await page.evaluate(f"() => {{ const el = document.querySelector('{sel}'); return el ? el.innerText.slice(0,150).replace(/\\s+/g,' ') : null }}")
                if cnt:
                    print(f"    [{sel}] {cnt[:120]}")
            # 表格?
            tbls = await page.evaluate("document.querySelectorAll('table').length")
            print(f"  tables: {tbls}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await page.wait_for_timeout(2000)

        # 2. ahhz 抓包: 监听所有 XHR/fetch 请求
        print("="*60)
        print("[ahhz] 抓包找 API")
        api_hits = []
        page.on("request", lambda req: api_hits.append((req.method, req.url)) if any(k in req.url for k in ['label', '8888', 'api', 'list', 'json', 'column', 'getList', 'Ajax']) else None)
        try:
            r = await page.goto("https://www.ahhz.gov.cn/zwgk/grassroots/column/6615882?&catId=1000260", timeout=45000, wait_until='networkidle')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(4000)
            print(f"  捕获请求 ({len(api_hits)}):")
            for m, u in api_hits[:20]:
                print(f"    {m} {u[:150]}")
            # 页面里含 8888/label 的脚本引用
            scripts = await page.eval_on_selector_all("script", "els => els.map(s => s.src || '').filter(s => s && (s.includes('8888') || s.includes('label') || s.includes('zwgk'))).slice(0,10)")
            for s in scripts:
                print(f"    script: {s[:150]}")
            # 列表区域 HTML 片段
            listhtml = await page.evaluate("() => { const el = document.querySelector('.list, .news-list, .column-list, .zwgk-list, ul.list, #list, .main-list'); return el ? el.outerHTML.slice(0, 800) : 'NO LIST EL' }")
            print(f"  列表容器: {listhtml[:500]}")
            # 页面中所有 li
            lis = await page.eval_on_selector_all("li", "els => els.slice(0,8).map(li => li.innerText.replace(/\\s+/g,' ').slice(0,70))")
            for li in lis:
                print(f"    li: {li}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await page.wait_for_timeout(2000)

        # 3. ahtxq 抓包
        print("="*60)
        print("[ahtxq] 抓包找 API")
        api_hits2 = []
        page.on("request", lambda req: api_hits2.append((req.method, req.url)) if any(k in req.url for k in ['label', '8888', 'api', 'list', 'json', 'column', 'getList', 'Ajax']) else None)
        try:
            r = await page.goto("https://www.ahtxq.gov.cn/zwgk/public/column/6615872?type=2&nav=0", timeout=45000, wait_until='networkidle')
            print(f"  goto: HTTP {r.status if r else 'None'}")
            await page.wait_for_timeout(4000)
            print(f"  捕获请求 ({len(api_hits2)}):")
            for m, u in api_hits2[:20]:
                print(f"    {m} {u[:150]}")
        except Exception as e:
            print(f"  ERROR: {e}")

        await browser.close()

asyncio.run(main())
