#!/usr/bin/env python3
"""探测 hanchuan: 分页 + 详情结构"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        await page.goto("http://www.hanchuan.gov.cn/tzgg/index.jhtml", wait_until='domcontentloaded', timeout=45000)
        await page.wait_for_timeout(6000)
        html = await page.content()
        # 分页: 找所有含页码/翻页的链接
        pagers = re.findall(r'<a[^>]*href="([^"]*)"[^>]*>\s*(下一页|末页|上一页|首页|\d+)\s*</a>', html)
        print(f"分页链接: {pagers}")
        # 找页码参数
        pg_params = re.findall(r'href="([^"]*(?:page|Page|index)[^"]*)"', html)
        print(f"page 参数: {list(dict.fromkeys(pg_params))[:10]}")
        # 列表容器 HTML
        lst = re.findall(r'<li[^>]*>([\s\S]{0,300}?)</li>', html)
        if lst:
            print(f"\n首个 li: {lst[0][:300]}")

        # 详情页
        print("="*60)
        print("[hanchuan] 详情页")
        det = "http://www.hanchuan.gov.cn/tzgg/2143292.jhtml"
        r = await page.goto(det, wait_until='domcontentloaded', timeout=35000)
        await page.wait_for_timeout(4000)
        dh = await page.content()
        print(f"status={r.status if r else '?'} title={(await page.title())[:60]} len={len(dh)}")
        metas = re.findall(r'<meta\s+name="(ArticleTitle|PubDate|ContentSource)"\s+content="([^"]*)"', dh)
        print(f"metas: {metas}")
        from collections import Counter
        divs = re.findall(r'<div[^>]*class="([^"]{3,60})"[^>]*>', dh)
        c = Counter(divs)
        print("\ndiv classes:")
        for cls, n in c.most_common(20):
            print(f"  div.{cls} x{n}")
        print("\n--- 长文本容器 ---")
        for cls in c:
            if c[cls] <= 2:
                idx = dh.find(f'class="{cls}"')
                if idx > 0:
                    start = dh.find('>', idx) + 1
                    text = re.sub(r'<[^>]+>', '', dh[start:start+300]).strip()
                    if len(text) > 40:
                        print(f"  [{cls}] {text[:70]}")
        print(f"\ntables: {len(re.findall(r'<table', dh))}")

        await browser.close()

asyncio.run(main())
