#!/usr/bin/env python3
"""探测 hjbh 详情页正文容器 (冷却后, 更大超时)"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 过盾: 首页 (重试)
        ok = False
        for attempt in range(3):
            try:
                r = await page.goto("http://cnqc.gov.cn/", wait_until='domcontentloaded', timeout=60000)
                await page.wait_for_timeout(15000)
                t = await page.title()
                print(f"[HOME try{attempt+1}] {r.status if r else '?'} {t[:40]}")
                if '请稍候' not in t:
                    ok = True
                    break
            except Exception as e:
                print(f"[HOME try{attempt+1}] ERR {str(e)[:60]}")
            await page.wait_for_timeout(15000)

        if not ok:
            print("过盾失败")
            await browser.close()
            return

        # 详情: 用一个已知 URL (关庄镇 生态宜居)
        url = "https://www.cnqc.gov.cn/gongkai/show/20220811145224789/20260810093929084.html"
        try:
            r = await page.goto(url, wait_until='domcontentloaded', timeout=60000)
            await page.wait_for_timeout(6000)
            dh = await page.content()
            print(f"[详情] status={r.status if r else '?'} len={len(dh)} title={(await page.title())[:60]}")
            from collections import Counter
            divs = re.findall(r'<div[^>]*class="([^"]{3,60})"[^>]*>', dh)
            c = Counter(divs)
            print("\ndiv classes:")
            for cls, n in c.most_common(30):
                print(f"  div.{cls} x{n}")
            print("\n--- 长文本容器 ---")
            for cls in c:
                if c[cls] == 1:
                    idx = dh.find(f'class="{cls}"')
                    if idx > 0:
                        start = dh.find('>', idx) + 1
                        text = re.sub(r'<[^>]+>', '', dh[start:start+400]).strip()
                        if len(text) > 40:
                            print(f"  [{cls}] {text[:70]}")
        except Exception as e:
            print(f"[详情 ERR] {str(e)[:80]}")

        await browser.close()

asyncio.run(main())
