#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
庐江县人民政府 - 公示公告 (www.lj.gov.cn/zwdt/gsgg/)
CMS: TRS (ul.doc_list > li > a[title][class=left] + span.right.date 列表,
     Ls.pagination JS -> /content/column/11245557?pageIndex=N 分页 API)
⚠️ 全站 JSL 反爬 (HTTP 521, __jsl_clearance_s cookie) → playwright-stealth 过盾
列表: ul.doc_list > li > a[href=/zwdt/gsgg/{数字ID}.html][title] + span.right.date
详情: meta ArticleTitle/PubDate/ContentSource 齐全
      正文 div#zoom.j-fontContent.newscontnet (深度计数闭合, 清理页脚/二维码/relInfo)
入库: crawler_lib.push_to_searchdb → /root/search.db (与 search_app 一致)
"""
import sys, os, re, asyncio, json, random, time, html as html_lib
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "https://www.lj.gov.cn"
LIST_PATH = "/zwdt/gsgg/"
LIST_URL = BASE_URL + LIST_PATH
COLUMN_ID = "11245557"
SITE_NAME = "庐江县人民政府-公示公告"
GROUP_NAME = "安徽"
SCRIPT_NAME = "crawl_lj_gsgg.py"
CUTOFF = "2023-08-17"

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

from playwright.async_api import async_playwright
from playwright_stealth import Stealth
import urllib.parse


def clean_title(t):
    """unescape entities + strip zero-width + collapse whitespace."""
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def parse_list(html_text):
    """Extract (url, title, date) from list page."""
    items = []
    # regex: li > a[title][href=.../数字.html] + span.right.date
    for m in re.finditer(
        r'<li[^>]*>\s*<a[^>]+href="([^"]*?/zwdt/gsgg/\d+\.html)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span[^>]*class="[^"]*date[^"]*"[^>]*>([^<]*)</span>',
        html_text, re.S,
    ):
        abs_url = m.group(1)
        if not abs_url.startswith("http"):
            abs_url = BASE_URL + abs_url
        title = clean_title(m.group(2))
        date = m.group(3).strip()
        if not re.match(r"^20\d{2}-\d{2}-\d{2}$", date):
            date = ""
        items.append({"url": abs_url, "title": title, "date": date})
    if not items:
        # fallback: 任意 .html 链接 + title 属性
        for m in re.finditer(
            r'<a[^>]+href="([^"]*?/\d+\.html)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span[^>]*class="[^"]*date[^"]*"[^>]*>([^<]*)</span>',
            html_text, re.S,
        ):
            abs_url = m.group(1)
            if not abs_url.startswith("http"):
                abs_url = BASE_URL + abs_url
            title = clean_title(m.group(2))
            date = m.group(3).strip()
            items.append({"url": abs_url, "title": title, "date": date})
    return items


def extract_zoom(html_text):
    """div 深度计数提取 #zoom 正文 (不依赖 clear 锚点, 防页脚吞入)."""
    body = ""
    m_open = re.search(r'<div[^>]*id="zoom"[^>]*>', html_text)
    if m_open:
        seg = html_text[m_open.end():]
        depth = 1
        for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
            if mm.group(0).startswith('</div>'):
                depth -= 1
                if depth == 0:
                    body = seg[:mm.start()]
                    break
            else:
                depth += 1
    if not body:
        # fallback: 无 zoom 时找 newscontnet/j-fontContent 容器
        m = re.search(r'<div[^>]*class="[^"]*(?:newscontnet|j-fontContent)[^"]*"[^>]*>', html_text)
        if m:
            seg = html_text[m.end():]
            depth = 1
            for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
                if mm.group(0).startswith('</div>'):
                    depth -= 1
                    if depth == 0:
                        body = seg[:mm.start()]
                        break
                else:
                    depth += 1
    return body


def clean_body(body):
    """清洗 zoom 正文: 脚本/样式/噪声块/页脚关键词截断."""
    body = re.sub(r'<script.*?</script>', '', body, flags=re.S)
    body = re.sub(r'<style.*?</style>', '', body, flags=re.S)

    def strip_block(html, pat):
        out = html
        while True:
            m = re.search(pat, out)
            if not m:
                break
            seg = out[m.end():]
            depth = 1
            end = m.end()
            for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
                if mm.group(0).startswith('</div>'):
                    depth -= 1
                    if depth == 0:
                        end = m.end() + mm.end()
                        break
                else:
                    depth += 1
            out = out[:m.start()] + out[end:]
        return out

    body = strip_block(body, r'<div[^>]*class="[^"]*wzewm[^"]*"[^>]*>')
    body = strip_block(body, r'<div[^>]*id="relInfo"[^>]*>')
    body = strip_block(body, r'<div[^>]*class="[^"]*fxd_close[^"]*"[^>]*>')
    body = re.sub(r'<!--.*?二维码.*?-->', '', body, flags=re.S)
    body = re.sub(r'<div[^>]*class="clear"[^>]*>\s*</div>', '', body, flags=re.S)
    body = re.sub(r'<p>\s*</p>', '', body)
    # 兜底: 尾部页脚关键词截断 (仅用不会出现在正文的词)
    for kw in ('中央政府网站', '省级政府网站', '市级政府网站', '关于我们', '主办单位：', '网站标识码', '本网站支持IPv6', '政务新媒体矩阵'):
        i = body.find(kw)
        if i > 0:
            body = body[:i]
    return body.strip()


def parse_detail(html_text, page_url):
    """Extract title, date, content from detail page."""
    title = ""
    date = ""
    m = re.search(r'name="ArticleTitle"\s+content="([^"]+)"', html_text)
    if not m:
        m = re.search(r'ArticleTitle"\s+content="([^"]+)"', html_text)
    if m:
        title = clean_title(m.group(1))
    m = re.search(r'name="PubDate"\s+content="([^"]+)"', html_text)
    if not m:
        m = re.search(r'PubDate"\s+content="([^"]+)"', html_text)
    if m:
        date = m.group(1).strip()[:10]
    body = extract_zoom(html_text)
    if not body:
        return None
    body = clean_body(body)
    if not body:
        return None
    return {"title": title, "pub_date": date, "body": body}


def html_to_text(content, page_url):
    """Convert zoom inner HTML to stored format:
    - keep <table> HTML intact
    - \\n\\n between <p> blocks
    - attachment <a> links embedded with absolute URL
    """
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    # 1) FIRST protect attachment <a> links (before table protection)
    content2 = content
    content2 = re.sub(
        r'<img[^>]*src="[^"]*fileTypeImages/icon_[a-z]+\.gif"[^>]*/?>', "", content2,
        flags=re.I,
    )
    attachments = []
    link_protect = {}

    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key

    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    # 2) THEN protect tables
    table_protect = []

    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"

    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)

    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # dedupe paragraphs
    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    return out, has_table, attachments


async def run_async(pages=5):
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled",
            "--no-sandbox",
        ])
        ctx = await browser.new_context(
            user_agent=UA,
            ignore_https_errors=True,
            viewport={"width": 1366, "height": 900},
        )
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. 过盾 (检查内容特征而非仅 title: WAF 挑战页 title 也是庐江)
        ok = False
        for i in range(5):
            try:
                resp = await page.goto(LIST_URL, wait_until="domcontentloaded", timeout=60000)
                await page.wait_for_timeout(3000)
                html_text = await page.content()
                if "doc_list" in html_text or "公示公告" in html_text:
                    ok = True
                    break
                print(f"  [WAF] attempt={i} status={resp.status if resp else '?'} 无列表特征, 重试")
            except Exception as e:
                print(f"  [WAF] goto retry {i}: {str(e)[:60]}")
            await page.wait_for_timeout(2500)
        if not ok:
            print("  [FATAL] 过盾失败")
            await browser.close()
            return 0

        # 2. 列表页 (fetch 直取, 不走 goto)
        items_all = []
        cutoff_hit = False
        total_pages = pages
        for pg in range(1, pages + 1):
            if pg == 1:
                url = LIST_URL
            else:
                url = f"{BASE_URL}/content/column/{COLUMN_ID}?pageIndex={pg}"
            try:
                html_text = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", url)
            except Exception as e:
                print(f"  [LIST-ERR] page={pg}: {e}")
                continue
            items = parse_list(html_text)
            print(f"  第{pg}页: {len(items)} 条")
            if not items:
                if pg > 1:
                    break
                continue
            m_pc = re.search(r'pageCount:(\d+)', html_text)
            if m_pc:
                total_pages = max(total_pages, int(m_pc.group(1)))
            for it in items:
                if it["date"] and it["date"] < CUTOFF:
                    cutoff_hit = True
                    print(f"  [CUTOFF] 第{pg}页出现 < {CUTOFF} 日期, 停止翻页")
                    break
                items_all.append(it)
            if cutoff_hit:
                break
            await page.wait_for_timeout(random.randint(1500, 3000))
            if pg >= total_pages:
                break

        print(f"  [LIST] 共收集 {len(items_all)} 条, total_pages={total_pages}")
        if not items_all:
            await browser.close()
            return 0

        # 3. 详情页
        records = []
        for i, it in enumerate(items_all, 1):
            url = it["url"]
            try:
                html_text = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", url)
            except Exception as e:
                print(f"  [{i}/{len(items_all)}] [ERR] {e}")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": it["date"],
                    "url": url, "content": f"<p>{it['title']}</p>",
                })
                continue
            d = parse_detail(html_text, url)
            if not d:
                print(f"  [{i}/{len(items_all)}] [NO-BODY] {it['title'][:40]}")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": it["date"],
                    "url": url, "content": f"<p>{it['title']}</p>",
                })
                continue
            title = d["title"] or it["title"]
            pub_date = d["pub_date"] or it["date"]
            content, has_table, attachments = html_to_text(d["body"], url)
            if len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                print(f"  [{i}/{len(items_all)}] [EMPTY] {title[:40]}")
                content = f"<p>{title}</p>"
            records.append({
                "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
                "url": url, "content": content, "attachments": json.dumps(attachments, ensure_ascii=False),
                "group_name": GROUP_NAME,
            })
            print(f"  [{i}/{len(items_all)}] 最新: {title[:60]}...")
            await page.wait_for_timeout(random.randint(1200, 2500))

        new_c = push_to_searchdb(records, batch_label="lj_gsgg")
        await browser.close()
        return new_c


if __name__ == "__main__":
    n = asyncio.run(run_async(pages=_PAGES))
    print(f"Done: {n} records")
