#!/usr/bin/env python3
"""探测 ahtxq / ahhz / zixing 三站: 过盾 + 列表结构 + API 同款性"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

SITES = {
    "ahtxq": "https://www.ahtxq.gov.cn/zxzx/ztzl/rdzt/hsjldtjjyq/qypd/index.html",
    "ahhz":  "https://www.ahhz.gov.cn/zwgk/grassroots/column/6615882?&catId=1000260",
    "zixing":"http://www.zixing.gov.cn/zwgk/ztbd/zwgkgzydzl/zlhms/hjbf/hjzl/default.htm",
}

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, viewport={'width':1366,'height':900})
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        for name, url in SITES.items():
            print(f"\n{'='*60}\n[{name}] {url}")
            try:
                resp = await page.goto(url, timeout=45000, wait_until='domcontentloaded')
                print(f"  goto: HTTP {resp.status if resp else 'None'}")
                await page.wait_for_timeout(3000)
                html = await page.content()
                print(f"  content size: {len(html)}")
                title = await page.title()
                print(f"  title: {title[:80]}")
                # 列表链接样本
                links = await page.eval_on_selector_all(
                    'a', 'els => els.slice(0,15).map(a => ({t:(a.innerText||"").trim().slice(0,50), h:a.href}))'
                )
                links = [l for l in links if l['t']]
                for l in links[:8]:
                    print(f"    link: [{l['t']}] {l['h'][:100]}")
                # 是否有 label/8888 同款 API 特征
                if '8888' in html or 'labelName' in html or 'publicInfoList' in html:
                    print("  >>> 含 label/8888 同款 API 特征!")
                if 'wzcon' in html or 'j-fontContent' in html:
                    print("  >>> 含 wzcon 正文类名特征!")
                # 提取可能的 siteId
                m = re.findall(r'siteId["\']?\s*[:=]\s*["\']?(\d{6,8})', html)
                if m: print(f"  siteId 候选: {set(m)}")
                # 提取 organId/catId 相关
                m2 = re.findall(r'organId["\']?\s*[:=]\s*["\']?(\d{6,8})', html)
                if m2: print(f"  organId 候选: {set(m2)}")
            except Exception as e:
                print(f"  ERROR: {e}")
            # 间隔
            await page.wait_for_timeout(2000)

        await browser.close()

asyncio.run(main())
