#!/usr/bin/env python3
"""探测 cnqc C: gongkai 详情页结构"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        await page.goto("http://cnqc.gov.cn/", wait_until='domcontentloaded', timeout=45000)
        await page.wait_for_timeout(12000)

        url = "https://www.cnqc.gov.cn/gongkai/show/20220811145224789/20260810093929084.html"
        print(f"[C详情] {url}")
        r = await page.goto(url, wait_until='domcontentloaded', timeout=40000)
        await page.wait_for_timeout(5000)
        print(f"  status={r.status if r else '?'} title={(await page.title())[:60]}")
        html = await page.content()
        metas = re.findall(r'<meta\s+name="(ArticleTitle|PubDate|ContentSource|PubDate)"\s+content="([^"]*)"', html)
        print(f"  metas: {metas}")
        # 正文容器
        for sel in ['msg-content', 'msg_content', 'content', 'article-content', 'wzcon', 'TRS_Editor', 'view-content']:
            idx = html.find(f'class="{sel}')
            if idx < 0:
                idx = html.find(f'class="{sel} ')
            if idx > 0:
                start = html.find('>', idx) + 1
                snippet = html[start:start+300]
                print(f"  [{sel}] {re.sub(r'<[^>]+>','',snippet)[:120]}")
        tbls = len(re.findall(r'<table', html))
        print(f"  tables: {tbls}")
        # 二维码位置 (div_div)
        qr = html.find('div_div')
        print(f"  div_div 位置: {qr}")

        await browser.close()

asyncio.run(main())
