#!/usr/bin/env python3
"""探测 www.ahshx.gov.cn 5个新栏目: breadcrumb栏目名 + API total"""
import re, asyncio, json

BASE = "https://www.ahshx.gov.cn"
SITE_ID = "6793339"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

# (organId, catId)
COLS = [
    ("6616285", "6719037"),
    ("6616369", "6719038"),
    ("6616268", "6719696"),
    ("6616370", "6719038"),
    ("6616368", "6727117"),
]

async def main():
    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 过盾: 第一个栏目列表页
        first = COLS[0]
        list_url = f"{BASE}/zwgk/public/column/{first[0]}?type=4&catId={first[1]}&action=list&nav=3"
        try:
            resp = await page.goto(list_url, wait_until='domcontentloaded', timeout=35000)
            await page.wait_for_timeout(10000)
            html = await page.content()
            m = re.search(r'<title>([^<]*)</title>', html)
            print(f"[LOAD] status={resp.status if resp else '?'} title={m.group(1) if m else '?'} len={len(html)}")
            if m and ('NWAF' in m.group(1) or 'Environment' in m.group(1)):
                print("[WAF] 仍被拦截")
                await browser.close()
                return
        except Exception as e:
            print(f"[LOAD ERR] {str(e)[:120]}")
            await browser.close()
            return

        for organ_id, cat_id in COLS:
            print(f"\n=== organId={organ_id} catId={cat_id} ===")
            # 1. 列表页 HTML 提取 breadcrumb
            list_url = f"{BASE}/zwgk/public/column/{organ_id}?type=4&catId={cat_id}&action=list&nav=3"
            try:
                html = await page.evaluate(f"""async () => {{
                    const r = await fetch('{list_url}', {{headers: {{'X-Requested-With': 'XMLHttpRequest'}}}});
                    return await r.text();
                }}""")
                # breadcrumb 找 位置> 后面链
                crumbs = re.findall(r'<a[^>]*href="[^"]*"[^>]*>([^<]{2,30})</a>', html)
                # 更精确: 找 class 含 position/crumb/daohang 的区域
                m = re.search(r'(?:class|id)="[^"]*(?:position|crumb|daohang|location)[^"]*"[^>]*>([\s\S]{0,2000})', html)
                crumb_text = ""
                if m:
                    seg = m.group(1)
                    crumb_text = re.sub(r'<[^>]+>', '>', seg)
                    crumb_text = re.sub(r'\s+', ' ', crumb_text).strip(' >')
                else:
                    # 兜底: 找 "> 首页 > xxx > xxx" 模式
                    m2 = re.search(r'首页\s*>\s*([^<]{1,100})', html)
                    if m2:
                        crumb_text = m2.group(1).strip()
                print(f"  breadcrumb: {crumb_text[:100]}")
            except Exception as e:
                print(f"  [HTML ERR] {str(e)[:100]}")

            # 2. API total
            api_url = (f"{BASE}/huangshanzwgk/zwgk/site/label/8888?_=0.{abs(hash(organ_id+cat_id))%10**15}"
                       f"&labelName=publicInfoList&siteId={SITE_ID}&organId={organ_id}"
                       f"&pageSize=15&pageIndex=1&isDate=true&dateFormat=yyyy-MM-dd&length=50"
                       f"&type=4&action=list&result=&isJson=true&keyWords=&isSetValue=true&catIds=&catId={cat_id}")
            try:
                data = await page.evaluate(f"""async () => {{
                    const r = await fetch('{api_url}');
                    return await r.json();
                }}""")
                total = data.get('total', 0)
                items = data.get('data', []) or []
                print(f"  API total={total}, page1={len(items)}")
                for it in items[:5]:
                    print(f"    - [{it.get('publishDate','')[:10]}] {it.get('title','')[:50]} | {it.get('link','')[:60]}")
            except Exception as e:
                print(f"  [API ERR] {str(e)[:120]}")

        await browser.close()

asyncio.run(main())
