#!/usr/bin/env python3
"""探测 cnqc 新栏目: 3个URL列表结构+名称+数量"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

URLS = [
    ("A", "https://www.cnqc.gov.cn/new/list/20171130181953989.html"),
    ("B", "https://cnqc.gov.cn/newList.aspx?classID=20171130181953989"),
    ("C", "https://www.cnqc.gov.cn/gongkai/list/20220811145224789.html?t=31"),
]

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 过盾: 首页
        try:
            resp = await page.goto("http://cnqc.gov.cn/", wait_until='domcontentloaded', timeout=45000)
            await page.wait_for_timeout(12000)
            t = await page.title()
            print(f"[HOME] status={resp.status if resp else '?'} title={t[:50]}")
        except Exception as e:
            print(f"[HOME ERR] {e}")
            await browser.close()
            return

        for tag, url in URLS:
            print(f"\n{'='*60}\n[{tag}] {url}")
            try:
                resp = await page.goto(url, wait_until='domcontentloaded', timeout=40000)
                await page.wait_for_timeout(6000)
                html = await page.content()
                t = await page.title()
                print(f"  status={resp.status if resp else '?'} title={t[:60]} len={len(html)}")
                # 列表项
                rows = re.findall(r'<li>\s*<a href="([^"]*\.html)"[^>]*title="([^"]*)"[^>]*>[\s\S]{0,200}?</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>', html)
                if not rows:
                    rows = re.findall(r'<a href="([^"]*Detail[^"]*\.html)"[^>]*title="([^"]*)"[^>]*>[\s\S]{0,200}?</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>', html)
                print(f"  列表项: {len(rows)}")
                for u, tt, d in rows[:5]:
                    print(f"    [{d}] {tt[:45]} -> {u[:80]}")
                # 分页
                pages = re.findall(r'page=(\d+)', html)
                print(f"  page 参数样本: {list(dict.fromkeys(pages))[:8]}")
                # total 文本
                tots = re.findall(r'共\s*(\d+)\s*条|(\d+)\s*条', html)
                print(f"  total 文本: {tots[:5]}")
            except Exception as e:
                print(f"  ERR: {str(e)[:100]}")
            await page.wait_for_timeout(2000)

        await browser.close()

asyncio.run(main())
