#!/usr/bin/env python3
"""探测 hngy: 列表结构+分页+详情 (复用 zixing 模式)"""
import asyncio, json, re, sys

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        stealth = Stealth()
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        url = "https://www.hngy.gov.cn/zwgk/zwdt/tzgg/default.htm"
        print(f"[hngy] {url}")
        try:
            r = await page.goto(url, wait_until='domcontentloaded', timeout=45000)
            await page.wait_for_timeout(6000)
            print(f"  status={r.status if r else '?'} title={(await page.title())[:60]}")
            html = await page.content()
            print(f"  len={len(html)}")
            # 列表项
            items = await page.evaluate("""() => {
                const out = [];
                document.querySelectorAll('ul li, .list li, li').forEach(li => {
                    const a = li.querySelector('a');
                    if (a && /content_\\d+|art_\\d+/.test(a.href)) {
                        const sp = li.querySelector('.time, span:last-child');
                        out.push({t: a.innerText.trim().slice(0,50), h: a.href, d: sp ? sp.innerText.trim() : '', html: li.outerHTML.slice(0,220)});
                    }
                });
                return out.slice(0,5);
            }""")
            for it in items:
                print(f"  item: [{it['d']}] {it['t']}")
                print(f"    {it['html']}")
            # 分页
            pager = await page.evaluate("""() => {
                const out = [];
                document.querySelectorAll('a').forEach(a => {
                    const t = (a.innerText||'').trim();
                    if (/^(下一页|末页|首页|上一页|\\d+)$/.test(t)) out.push({t, h: a.href});
                });
                return out.slice(0,15);
            }""")
            for pg in pager:
                print(f"  pager: [{pg['t']}] {pg['h'][:90]}")
            tots = await page.evaluate("document.body.innerText.match(/共\\s*\\d+\\s*条|\\d+\\s*条/g)")
            print(f"  total: {tots[:3] if tots else 'None'}")
        except Exception as e:
            print(f"  ERR: {str(e)[:100]}")

        # 详情页 (如果有)
        await page.wait_for_timeout(1500)
        print("="*60)
        print("[hngy] 详情页")
        try:
            # 从列表拿第一个详情 URL
            det = await page.evaluate("""() => {
                const a = document.querySelector('li a[href*="content_"], li a[href*="art_"]');
                return a ? a.href : null;
            }""")
            if det:
                print(f"  详情 URL: {det}")
                r = await page.goto(det, wait_until='domcontentloaded', timeout=35000)
                await page.wait_for_timeout(4000)
                print(f"  status={r.status if r else '?'} title={(await page.title())[:60]}")
                dh = await page.content()
                metas = re.findall(r'<meta\s+name="(ArticleTitle|PubDate|ContentSource)"\s+content="([^"]*)"', dh)
                print(f"  metas: {metas}")
                for sel in ['content', 'article-content', 'TRS_Editor', 'zwxl', 'view', 'xl_content', 'wzcon']:
                    idx = dh.find(f'class="{sel}')
                    if idx > 0:
                        start = dh.find('>', idx) + 1
                        snippet = re.sub(r'<[^>]+>', '', dh[start:start+300])[:120]
                        print(f"  [{sel}] {snippet}")
                print(f"  tables: {len(re.findall(r'<table', dh))}")
            else:
                print("  无详情链接")
        except Exception as e:
            print(f"  ERR: {str(e)[:100]}")

        await browser.close()

asyncio.run(main())
