#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_ahtxq_qypd.py - 黄山市屯溪区-营商环境企业频道 (ahtxq.gov.cn)
站点: 帝联NWAF + 瑞数4代双WAF (1eeb27c + ctct_bundle) -> 必须 playwright stealth
列表: /zxzx/ztzl/rdzt/hsjldtjjyq/qypd/index.html (静态, 无分页, ~9条)
详情: /zxzx/ztzl/rdzt/hsjldtjjyq/qypd/{id}.html
  标题: h1 / meta ArticleTitle | 日期: meta PubDate / 页面文本 | 正文: 待确认
"""
import sys, os, re, asyncio, json, random, time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "屯溪区-营商环境企业频道"
BASE = "https://www.ahtxq.gov.cn"
LIST_URL = BASE + "/zxzx/ztzl/rdzt/hsjldtjjyq/qypd/index.html"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

from playwright.async_api import async_playwright
from playwright_stealth import Stealth


async def goto_with_retry(page, url, timeout=45000, retries=3):
    for attempt in range(retries):
        try:
            resp = await page.goto(url, wait_until='domcontentloaded', timeout=timeout)
            await page.wait_for_timeout(5000)
            return resp
        except Exception:
            if attempt < retries - 1:
                await page.wait_for_timeout(3000)
    return None


async def run_async(pages=5):
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled",
            "--no-sandbox",
        ])
        ctx = await browser.new_context(
            user_agent=UA,
            ignore_https_errors=True,
            viewport={"width": 1366, "height": 900},
        )
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        resp = await goto_with_retry(page, LIST_URL)
        print(f"[LOAD] status={resp.status if resp else 'N/A'} title={await page.title()}")

        # 列表: 所有 qypd 详情链接
        items = await page.evaluate("""() => {
            const out = [];
            document.querySelectorAll('a[href*="/qypd/"].html, a[href*="/qypd/"]').forEach(a => {
                const href = a.getAttribute('href') || '';
                const t = (a.innerText || '').trim().replace(/\\s+/g, ' ');
                if (href.includes('/qypd/') && /\\d+\\.html$/.test(href) && t.length > 5) {
                    if (!out.find(x => x.href === href)) out.push({t: t, href: href});
                }
            });
            return out;
        }""")
        # 绝对化 + 去重
        seen = set()
        items2 = []
        for it in items:
            href = it['href']
            if href.startswith('/'):
                href = BASE + href
            elif not href.startswith('http'):
                continue
            if href in seen:
                continue
            seen.add(href)
            items2.append({"title": it['t'], "url": href})
        print(f"[LIST] {len(items2)} 条")
        if not items2:
            await browser.close()
            return 0

        # 详情页
        records = []
        for i, it in enumerate(items2, 1):
            url = it["url"]
            resp = await goto_with_retry(page, url)
            html = await page.content()

            # 标题
            title = it["title"]
            m = re.search(r'name="ArticleTitle"\s+content="([^"]+)"', html)
            if not m:
                m = re.search(r'<h1[^>]*>([^<]+)</h1>', html)
            if m:
                t2 = m.group(1).strip()
                if t2 and len(t2) > 3:
                    title = t2

            # 日期
            pub_date = ""
            m = re.search(r'name="PubDate"\s+content="([^"]+)"', html)
            if m:
                pub_date = m.group(1).strip()[:10]
            if not pub_date:
                m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
                if m:
                    pub_date = m.group(1)

            # 正文容器
            body = ""
            for sel_pat in [r'<div[^>]*class="[^"]*wzcon[^"]*"[^>]*>(.*?)</div>\s*<div[^>]*class="[^"]*clear[^"]*"',
                            r'<div[^>]*class="[^"]*content[^"]*"[^>]*>(.*?)</div>\s*<div[^>]*class="[^"]*clear[^"]*"',
                            r'<div[^>]*id="zoom"[^>]*>(.*?)</div>']:
                m = re.search(sel_pat, html, re.S)
                if m:
                    body = m.group(1)
                    break
            if not body:
                # 尝试最长 div
                m = re.search(r'<div[^>]*class="[^"]*(?:article|news|detail|content)[^"]*"[^>]*>(.*)', html, re.S)
                if m:
                    body = m.group(1)[:8000]

            if body:
                body = re.sub(r'<script.*?</script>', '', body, flags=re.S)
                body = re.sub(r'<style.*?</style>', '', body, flags=re.S)
                body = re.sub(r'href="/', 'href="' + BASE + '/', body)
                body = re.sub(r'src="/', 'src="' + BASE + '/', body)
                body = body.strip()
                records.append({
                    "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
                    "content": body, "source_url": url, "url": url,
                })
            else:
                print(f"  [{i}/{len(items2)}] [SKIP-EMPTY] {title[:50]}...")
                records.append({
                    "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
                    "content": f"<p>{title}</p>", "source_url": url, "url": url,
                })
            print(f"  [{i}/{len(items2)}] {title[:50]}... (body={len(body)})")
            await page.wait_for_timeout(random.randint(1500, 3000))

        await browser.close()
        new_c = push_to_searchdb(records, batch_label="ahtxq_qypd")
        return new_c


def main():
    pages = 5
    for i, a in enumerate(sys.argv):
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
    n = asyncio.run(run_async(pages=pages))
    print(f"Done: {n} records")


if __name__ == "__main__":
    main()
