#!/usr/bin/env python3
"""探测 hanchuan: 正文容器确认 + 分页方式"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 详情正文
        det = "http://www.hanchuan.gov.cn/tzgg/2143292.jhtml"
        await page.goto(det, wait_until='domcontentloaded', timeout=35000)
        await page.wait_for_timeout(4000)
        dh = await page.content()
        for sel in ['p-xwxq-content', 'p-hcgk-right-content-dy', 'p-hcgk-right-content-title']:
            idx = dh.find(f'class="{sel}')
            if idx > 0:
                start = dh.find('>', idx) + 1
                text = re.sub(r'<[^>]+>', '', dh[start:start+400]).strip()
                print(f"[{sel}] {text[:100]}")
        # 附件
        atts = re.findall(r'<a[^>]*href="([^"]*\.(?:pdf|docx?|xlsx?|zip|rar|wps))"[^>]*>([^<]*)</a>', dh)
        print(f"附件: {atts[:5]}")
        tbls = len(re.findall(r'<table', dh))
        print(f"tables: {tbls}")

        # 分页测试
        print("="*60)
        for pg_url in [
            "http://www.hanchuan.gov.cn/tzgg/index_2.jhtml",
            "http://www.hanchuan.gov.cn/tzgg/index.jhtml?page=2",
            "http://www.hanchuan.gov.cn/tzgg/index_1.jhtml",
        ]:
            try:
                r = await page.goto(pg_url, wait_until='domcontentloaded', timeout=25000)
                await page.wait_for_timeout(2500)
                t = await page.title()
                items = await page.evaluate("""() => {
                    const out = [];
                    document.querySelectorAll('li a').forEach(a => {
                        const t = (a.innerText||'').trim();
                        if (t && t.length > 8 && !t.startsWith('首页')) out.push(t.slice(0,30));
                    });
                    return out.slice(0,3);
                }""")
                print(f"[{pg_url}] {r.status if r else '?'} title={t[:35]} items={items}")
            except Exception as e:
                print(f"[{pg_url}] ERR {str(e)[:50]}")

        await browser.close()

asyncio.run(main())
