#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_lj_hpgs.py - 庐江县人民政府-建设项目环评文件（报告书、报告表）审批 (lj.gov.cn)
站点: 合肥政务公开站群 (hefeizwgk) + WAF __jsl_clearance 521 挑战 -> 必须 playwright 过盾
列表: GET /mayor/site/label/8888?labelName=publicInfoList&siteId=6784341&organId=19081
      &catId=7038868&pageSize=20&pageIndex={N}&isDate=true&dateFormat=yyyy-MM-dd
      &length=50&type=4&action=list  (返回 HTML 片段, pageCount=26 页 x20 = ~520 条)
列表条目: div.xxgk_navli > ul > li.mc > a (标题+href) + li.rq (日期) + li.yh (索引号)
详情: /public/19081/{id}.html
  标题: meta ArticleTitle | 日期: meta PubDate (取前10) | 来源: meta ContentSource
  正文: div#zoom.j-fontContent
内容: 环评受理/拟审批/批复公示 (庐江县生态环境分局)
"""
import sys, os, re, asyncio, json, random, time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "庐江县人民政府-建设项目环评审批"  # 沿用旧 site_name, 新旧数据按 URL 去重合并
BASE = "https://www.lj.gov.cn"
SITE_ID = "6784341"
ORGAN_ID = "19081"
CAT_ID = "7038868"
PAGE_SIZE = 20
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

CUTOFF = "2023-08-17"  # 3 年截断

from playwright.async_api import async_playwright
from playwright_stealth import Stealth


def list_api(page_idx):
    return (f"{BASE}/mayor/site/label/8888?labelName=publicInfoList&siteId={SITE_ID}"
            f"&organId={ORGAN_ID}&catId={CAT_ID}&pageSize={PAGE_SIZE}&pageIndex={page_idx}"
            f"&isDate=true&dateFormat=yyyy-MM-dd&length=50&type=4&action=list")


def parse_list(html):
    """解析 API 返回的 HTML 片段"""
    items = []
    # 每个条目: div.xxgk_navli ... </ul> (块内嵌套 clear div, 以 </ul> 结束)
    for block in re.findall(r'<div class="xxgk_navli">(.*?)</ul>', html, re.S):
        m_title = re.search(r'<li class="mc">\s*<a[^>]*href="([^"]+)"[^>]*>\s*(.*?)\s*</a>', block, re.S)
        m_date = re.search(r'<li class="rq">([^<]+)</li>', block)
        m_idx = re.search(r'<li class="yh">([^<]+)</li>', block)
        if not m_title:
            continue
        href = m_title.group(1)
        title = re.sub(r'\s+', ' ', m_title.group(2)).strip()
        title = re.sub(r'^[•·]\s*', '', title).strip()
        if not title or not href.startswith('http'):
            continue
        items.append({
            "title": title,
            "url": href,
            "date": m_date.group(1).strip() if m_date else "",
            "index_no": m_idx.group(1).strip() if m_idx else "",
        })
    return items


def parse_detail(html):
    """详情页解析: meta + div#zoom 正文"""
    title = ""
    m = re.search(r'name="ArticleTitle"\s+content="([^"]+)"', html)
    if not m:
        m = re.search(r'ArticleTitle"\s+content="([^"]+)"', html)
    if m:
        title = m.group(1).strip()

    pub_date = ""
    m = re.search(r'name="PubDate"\s+content="([^"]+)"', html)
    if not m:
        m = re.search(r'PubDate"\s+content="([^"]+)"', html)
    if m:
        pub_date = m.group(1).strip()[:10]

    source = ""
    m = re.search(r'name="ContentSource"\s+content="([^"]+)"', html)
    if not m:
        m = re.search(r'ContentSource"\s+content="([^"]+)"', html)
    if m:
        source = m.group(1).strip()

    # 正文容器 div#zoom — 用 div 深度计数找真正闭合点(不能依赖 clear 锚点,
    # 否则 zoom 后跟 wzewm 二维码时 fallback 贪婪匹配会把整页页脚吞进正文)
    body = ""
    m_open = re.search(r'<div[^>]*id="zoom"[^>]*>', html)
    if m_open:
        seg = html[m_open.end():]
        depth = 1
        i = 0
        for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
            if mm.group(0).startswith('</div>'):
                depth -= 1
                if depth == 0:
                    body = seg[:mm.start()]
                    break
            else:
                depth += 1
    if not body:
        return None

    # 清洗: 移除脚本/样式/隐藏区
    body = re.sub(r'<script.*?</script>', '', body, flags=re.S)
    body = re.sub(r'<style.*?</style>', '', body, flags=re.S)

    # 按 div 深度移除完整嵌套块 (wzewm 二维码 / relInfo / fxd_close 等)
    def strip_block(html, pat):
        out = html
        while True:
            m = re.search(pat, out)
            if not m:
                break
            seg = out[m.end():]
            depth = 1
            end = m.end()
            for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
                if mm.group(0).startswith('</div>'):
                    depth -= 1
                    if depth == 0:
                        end = m.end() + mm.end()
                        break
                else:
                    depth += 1
            out = out[:m.start()] + out[end:]
        return out

    body = strip_block(body, r'<div[^>]*class="[^"]*wzewm[^"]*"[^>]*>')
    body = strip_block(body, r'<div[^>]*id="relInfo"[^>]*>')
    body = strip_block(body, r'<div[^>]*class="[^"]*fxd_close[^"]*"[^>]*>')
    body = re.sub(r'<!--.*?二维码.*?-->', '', body, flags=re.S)
    body = re.sub(r'<div[^>]*class="clear"[^>]*>\s*</div>', '', body, flags=re.S)
    body = re.sub(r'<p>\s*</p>', '', body)
    # 兜底: 尾部页脚关键词截断 (仅用不会出现在正文的词, 勿用 关闭/打印 等)
    for kw in ('中央政府网站', '省级政府网站', '市级政府网站', '关于我们', '主办单位：', '网站标识码', '本网站支持IPv6', '政务新媒体矩阵'):
        i = body.find(kw)
        if i > 0:
            body = body[:i]
    # 附件链接绝对化
    def abs_url(match):
        href = match.group(1)
        if href.startswith('//'):
            href = 'https:' + href
        elif href.startswith('/'):
            href = BASE + href
        return f'<a href="{href}"'
    body = re.sub(r'<a[^>]*href="([^"]+)"', abs_url, body)
    # 微信分享/打印等工具条
    body = re.sub(r'<div[^>]*class="[^"]*(share|weixin|ewm|print|scan)[^"]*"[^>]*>.*?</div>', '', body, flags=re.S)
    # 空段落清理
    body = re.sub(r'<p>\s*</p>', '', body)
    body = body.strip()

    return {"title": title, "pub_date": pub_date, "source": source, "body": body}


async def run_async(pages=5):
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled",
            "--no-sandbox",
        ])
        ctx = await browser.new_context(
            user_agent=UA,
            ignore_https_errors=True,
            viewport={"width": 1366, "height": 900},
        )
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. 过盾
        ok = False
        for attempt in range(3):
            try:
                resp = await page.goto(f"{BASE}/public/column/19081?type=4&catId={CAT_ID}&action=list", wait_until="domcontentloaded", timeout=60000)
                await page.wait_for_timeout(8000)
                title = await page.title()
                print(f"[LOAD] attempt={attempt+1} status={resp.status if resp else 'N/A'} title={title}")
                ok = True
                break
            except Exception as e:
                print(f"[LOAD-ERR] attempt={attempt+1}: {e}")
                await page.wait_for_timeout(5000)
        if not ok:
            await browser.close()
            return 0

        items_all = []
        cutoff_hit = False
        total_pages = pages

        for pg in range(1, pages + 1):
            api = list_api(pg)
            try:
                html = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", api)
            except Exception as e:
                print(f"  [API-ERR] page={pg}: {e}")
                continue
            items = parse_list(html)
            print(f"  第{pg}页: {len(items)} 条")
            if not items:
                # 可能到末页
                if pg > 1:
                    break
                continue
            # 提取 pageCount (从 API 尾部 Ls.pagination)
            m_pc = re.search(r'pageCount:(\d+)', html)
            if m_pc:
                total_pages = max(total_pages, int(m_pc.group(1)))
            for it in items:
                # CUTOFF 检查
                if it["date"] and it["date"] < CUTOFF:
                    cutoff_hit = True
                    break
                items_all.append(it)
            if cutoff_hit:
                print(f"  [CUTOFF] 第{pg}页出现 < {CUTOFF} 日期, 停止翻页")
                break
            await page.wait_for_timeout(random.randint(2000, 4000))
            # 用 pageCount 停止
            if pg >= total_pages:
                break

        print(f"  [LIST] 共收集 {len(items_all)} 条, total_pages={total_pages}")
        if not items_all:
            await browser.close()
            return 0

        # 2. 详情页
        records = []
        for i, it in enumerate(items_all, 1):
            url = it["url"]
            try:
                html = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", url)
            except Exception as e:
                print(f"  [{i}/{len(items_all)}] [ERR] {e}")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": it["date"],
                    "content": f"<p>{it['title']}</p>", "source_url": url, "url": url,
                })
                continue
            d = parse_detail(html)
            if d and d["body"]:
                title = d["title"] or it["title"]
                content = d["body"]
                pub_date = d["pub_date"] or it["date"]
                records.append({
                    "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
                    "content": content, "source_url": url, "url": url,
                })
            else:
                print(f"  [{i}/{len(items_all)}] [SKIP-EMPTY] {it['title'][:50]}...")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": it["date"],
                    "content": f"<p>{it['title']}</p>", "source_url": url, "url": url,
                })
            if i % 10 == 0:
                print(f"  [{i}/{len(items_all)}] 最新: {it['title'][:60]}...")
            await page.wait_for_timeout(random.randint(1500, 3000))

        await browser.close()

        # 3. 入库
        new_c = push_to_searchdb(records, batch_label="lj_hpgs")
        return new_c


def main():
    pages = 5
    for i, a in enumerate(sys.argv):
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
    n = asyncio.run(run_async(pages=pages))
    print(f"Done: {n} records")


if __name__ == "__main__":
    main()
