#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_jinxiang_tzgg.py - 金乡县通知公告 (jinxiang.gov.cn col18311, JXA61)
站点: 无 WAF, 但 curl 被本地 TLS 拦截 -> playwright 稳定
列表: POST /module/xxgk/search.jsp
  body: infotypeId=JXA61&jdid=96&area=&divid=div4&vc_title=&vc_number=
        &sortfield=compaltedate:0&currpage={N}&standardXxgk=1&isAllList=1
  返回 HTML 片段, 每页 15 条, total=3102 (207页)
列表条目: a[href*="art_18311_"] 详情链接
详情: /art/YYYY/M/D/art_18311_{id}.html?xxgkhide=1
  标题: meta ArticleTitle | 来源: meta ContentSource | 正文: div#zoom
"""
import sys, os, re, asyncio, json, random, time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "金乡县-通知公告"
BASE = "http://www.jinxiang.gov.cn"
SEARCH_URL = BASE + "/module/xxgk/search.jsp?standardXxgk=1&isAllList=1&currpage={p}&sortfield=compaltedate:0"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

CUTOFF = "2023-08-17"

from playwright.async_api import async_playwright
from playwright_stealth import Stealth


def search_body(currpage):
    return {
        "infotypeId": "JXA61",
        "jdid": "96",
        "area": "",
        "divid": "div4",
        "vc_title": "",
        "vc_number": "",
        "sortfield": "compaltedate:0",
        "currpage": str(currpage),
        "vc_filenumber": "",
        "vc_all": "",
        "texttype": "",
        "fbtime": "",
        "standardXxgk": "1",
        "isAllList": "1",
    }


def parse_list(html):
    """解析 search.jsp 返回的 HTML 片段"""
    items = []
    # 条目链接: art_18311_xxx.html
    for m in re.finditer(r'<a[^>]*href="([^"]*art_18311_\d+\.html[^"]*)"[^>]*>(.*?)</a>', html, re.S):
        href = m.group(1)
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        title = re.sub(r'\s+', ' ', title).strip()
        title = re.sub(r'^【[^】]*】', '', title).strip()
        if not href.startswith('http'):
            href = BASE + href
        items.append({"title": title, "url": href})
    # 去重
    seen = set()
    out = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            out.append(it)
    return out


def parse_detail(html):
    title = ""
    m = re.search(r'name="ArticleTitle"\s+content="([^"]+)"', html)
    if m:
        title = m.group(1).strip()

    pub_date = ""
    m = re.search(r'name="pubdate"\s+content="([^"]+)"', html)
    if m:
        pub_date = m.group(1).strip()[:10]

    source = ""
    m = re.search(r'name="ContentSource"\s+content="([^"]+)"', html)
    if m:
        source = m.group(1).strip()

    # 正文 div#zoom
    body = ""
    m = re.search(r'<div[^>]*id="zoom"[^>]*>(.*?)</div>\s*<div[^>]*class="[^"]*(?:clear|foot)[^"]*"', html, re.S)
    if not m:
        m = re.search(r'<div[^>]*id="zoom"[^>]*>(.*)', html, re.S)
    if m:
        body = m.group(1)
    if not body:
        return None

    body = re.sub(r'<script.*?</script>', '', body, flags=re.S)
    body = re.sub(r'<style.*?</style>', '', body, flags=re.S)
    body = re.sub(r'<!--.*?-->', '', body, flags=re.S)
    # 去掉模板注释
    body = re.sub(r'<!--\$\[信息内容\]-->', '', body)
    body = re.sub(r'<!--ZJEG_RSS\.content\.(?:begin|end)-->', '', body)
    body = re.sub(r'<\$\[信息内容\]>', '', body)

    def abs_url(match):
        href = match.group(1)
        if href.startswith('//'):
            href = 'http:' + href
        elif href.startswith('/'):
            href = BASE + href
        return f'<a href="{href}"'
    body = re.sub(r'<a[^>]*href="([^"]+)"', abs_url, body)
    body = re.sub(r'<p>\s*</p>', '', body)
    body = body.strip()
    return {"title": title, "source": source, "body": body, "pub_date": pub_date}


async def run_async(pages=5):
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled",
            "--no-sandbox",
        ])
        ctx = await browser.new_context(
            user_agent=UA,
            ignore_https_errors=True,
            viewport={"width": 1366, "height": 900},
        )
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 过盾: 栏目首页
        ok = False
        for attempt in range(3):
            try:
                resp = await page.goto(f"{BASE}/col/col18311/index.html?vc_xxgkarea=jnsjxx&number=JXA61&jh=263", wait_until="domcontentloaded", timeout=60000)
                await page.wait_for_timeout(8000)
                print(f"[LOAD] attempt={attempt+1} status={resp.status if resp else 'N/A'} title={await page.title()}")
                ok = True
                break
            except Exception as e:
                print(f"[LOAD-ERR] {attempt+1}: {str(e)[:60]}")
                await page.wait_for_timeout(5000)
        if not ok:
            await browser.close()
            return 0

        # 列表页 (search.jsp POST)
        items_all = []
        cutoff_hit = False
        for pg in range(1, pages + 1):
            try:
                result = await page.evaluate("""async (args) => {
                    const form = new URLSearchParams();
                    for (const [k, v] of Object.entries(args.body)) form.append(k, v);
                    const res = await fetch(args.url, {
                        method: 'POST',
                        body: form,
                        credentials: 'include',
                        headers: {'Content-Type': 'application/x-www-form-urlencoded'}
                    });
                    return {status: res.status, body: await res.text()};
                }""", {"url": SEARCH_URL.format(p=pg), "body": search_body(pg)})
                html = result["body"]
            except Exception as e:
                print(f"  [API-ERR] page={pg}: {str(e)[:80]}")
                continue
            items = parse_list(html)
            print(f"  第{pg}页: {len(items)} 条")
            if not items:
                break
            for it in items:
                items_all.append(it)
            # 找日期判断 CUTOFF (列表 HTML 里可能有日期)
            dates = re.findall(r'(\d{4}-\d{2}-\d{2})', html)
            if dates and min(dates) < CUTOFF:
                cutoff_hit = True
                print(f"  [CUTOFF] 第{pg}页出现 < {CUTOFF}")
                break
            await page.wait_for_timeout(random.randint(2000, 3500))

        print(f"  [LIST] 共 {len(items_all)} 条")
        if not items_all:
            await browser.close()
            return 0

        # 详情页
        records = []
        for i, it in enumerate(items_all, 1):
            url = it["url"]
            try:
                dhtml = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", url)
            except Exception as e:
                print(f"  [{i}/{len(items_all)}] [ERR] {str(e)[:60]}")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": "",
                    "content": f"<p>{it['title']}</p>", "source_url": url, "url": url,
                })
                continue
            d = parse_detail(dhtml)
            if d and d["body"]:
                records.append({
                    "site_name": SITE_NAME, "title": d["title"] or it["title"], "pub_date": d.get("pub_date", ""),
                    "content": d["body"], "source_url": url, "url": url,
                })
            else:
                print(f"  [{i}/{len(items_all)}] [SKIP-EMPTY] {it['title'][:50]}...")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": "",
                    "content": f"<p>{it['title']}</p>", "source_url": url, "url": url,
                })
            if i % 15 == 0:
                print(f"  [{i}/{len(items_all)}] ...")
            await page.wait_for_timeout(random.randint(1500, 3000))

        await browser.close()
        new_c = push_to_searchdb(records, batch_label="jinxiang_tzgg")
        return new_c


def main():
    pages = 5
    for i, a in enumerate(sys.argv):
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
    n = asyncio.run(run_async(pages=pages))
    print(f"Done: {n} records")


if __name__ == "__main__":
    main()
