#!/usr/bin/env python3
"""探测 cnqc C: gongkai 环境保护列表结构"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        await page.goto("http://cnqc.gov.cn/", wait_until='domcontentloaded', timeout=45000)
        await page.wait_for_timeout(12000)

        print("="*60)
        print("[C] gongkai 环境保护列表")
        await page.goto("https://www.cnqc.gov.cn/gongkai/list/20220811145224789.html?t=31", wait_until='domcontentloaded', timeout=40000)
        await page.wait_for_timeout(6000)
        html = await page.content()
        # 找列表容器
        m = re.search(r'<ul[^>]*class="[^"]*list[^"]*"[^>]*>([\s\S]*?)</ul>', html)
        if not m:
            m = re.search(r'<div[^>]*class="[^"]*list[^"]*"[^>]*>([\s\S]*?)</div>', html)
        if m:
            print(f"  列表容器: {m.group(1)[:800]}")
        else:
            print("  无标准列表容器, 找所有含 Detail 链接:")
            dets = re.findall(r'<a[^>]*href="([^"]*Detail[^"]*\.html)"[^>]*>([\s\S]{0,80}?)</a>', html)
            for u, t in dets[:8]:
                print(f"    {u[:70]} | {re.sub(r'<[^>]+>','',t).strip()[:40]}")
        # 所有链接标题
        links = re.findall(r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"', html)
        print(f"\n  总链接数: {len(links)}")
        for u, t in links[:10]:
            if 'Detail' in u or 'detail' in u:
                print(f"    [{t[:45]}] {u[:80]}")
        # 分页结构
        pg = re.findall(r'href="([^"]*page=\d+[^"]*)"[^>]*>(\d+)</a>', html)
        print(f"\n  分页: {pg[:10]}")

        await browser.close()

asyncio.run(main())
