#!/usr/bin/env python3
"""探测 hanchuan: 列表 li 原始 HTML"""
import asyncio, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        await page.goto("http://www.hanchuan.gov.cn/tzgg/index.jhtml", wait_until='domcontentloaded', timeout=40000)
        await page.wait_for_timeout(6000)
        html = await page.content()
        # 找包含 tzgg/ 数字链接的 li 或容器
        idx = html.find('tzgg/2143292')
        if idx > 0:
            start = max(0, idx - 500)
            print(f"--- 2143292 附近 ---")
            print(html[start:idx+300])
        else:
            # 找所有 li
            lis = re.findall(r'<li[^>]*>([\s\S]{0,400}?)</li>', html)
            print(f"li 数量: {len(lis)}")
            for li in lis[:5]:
                if 'tzgg' in li or 'jhtml' in li:
                    print(f"--- li ---")
                    print(li[:350])
                    print()

        await browser.close()

asyncio.run(main())
