#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
旌德县人民政府 - 建设项目环评文件（报告书、报告表）审批 (www.ahjd.gov.cn)
列表: /XxgkContent/showList/814/125539/page_{N}.html
      ul.search-list.qhlist > li > a[href=/OpennessContent/show/{id}.html][title] + span[YYYY-MM-DD]
详情: /OpennessContent/show/{id}.html
      标题: meta ArticleTitle | 日期: meta PubDate(取前10) | 来源: meta ContentSource
      正文: div#zoom (深度计数闭合, 清理页脚/二维码/relInfo)
入库: crawler_lib.push_to_searchdb → /root/search.db
"""
import sys, os, re, asyncio, json, random, time, urllib.parse
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "旌德县-建设项目环评审批"
BASE = "https://www.ahjd.gov.cn"
COLUMN_ID = "125539"
CUTOFF = "2023-08-17"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

from playwright.async_api import async_playwright
from playwright_stealth import Stealth


def list_url(page):
    return f"{BASE}/XxgkContent/showList/814/{COLUMN_ID}/page_{page}.html"


def parse_list(html_text):
    items = []
    for m in re.finditer(
        r'<a[^>]+href="(/OpennessContent/show/\d+\.html)"[^>]*title="([^"]*)"[^>]*>',
        html_text, re.S,
    ):
        href, title = m.group(1), m.group(2).strip()
        title = re.sub(r"\s+", " ", title)
        # 日期: a 后面 300 字符内找 span 日期
        date = ""
        m_date = re.search(r'<span>\s*(\d{4}-\d{2}-\d{2})\s*</span>', html_text[m.end():m.end()+400])
        if m_date:
            date = m_date.group(1)
        if not title or not re.match(r"^20\d{2}-\d{2}-\d{2}$", date):
            continue
        items.append({"url": BASE + href, "title": title, "date": date})
    return items


def extract_zoom(html_text):
    body = ""
    m_open = re.search(r'<div[^>]*id="zoom"[^>]*>', html_text)
    if m_open:
        seg = html_text[m_open.end():]
        depth = 1
        for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
            if mm.group(0).startswith('</div>'):
                depth -= 1
                if depth == 0:
                    body = seg[:mm.start()]
                    break
            else:
                depth += 1
    return body


def clean_body(body):
    body = re.sub(r'<script.*?</script>', '', body, flags=re.S)
    body = re.sub(r'<style.*?</style>', '', body, flags=re.S)

    def strip_block(html, pat):
        out = html
        while True:
            m = re.search(pat, out)
            if not m:
                break
            seg = out[m.end():]
            depth = 1
            end = m.end()
            for mm in re.finditer(r'<div\b[^>]*>|</div>', seg):
                if mm.group(0).startswith('</div>'):
                    depth -= 1
                    if depth == 0:
                        end = m.end() + mm.end()
                        break
                else:
                    depth += 1
            out = out[:m.start()] + out[end:]
        return out

    body = strip_block(body, r'<div[^>]*class="[^"]*wzewm[^"]*"[^>]*>')
    body = strip_block(body, r'<div[^>]*id="relInfo"[^>]*>')
    body = strip_block(body, r'<div[^>]*class="[^"]*fxd_close[^"]*"[^>]*>')
    body = re.sub(r'<!--.*?二维码.*?-->', '', body, flags=re.S)
    body = re.sub(r'<div[^>]*class="clear"[^>]*>\s*</div>', '', body, flags=re.S)
    body = re.sub(r'<p>\s*</p>', '', body)
    for kw in ('中央政府网站', '省级政府网站', '市级政府网站', '关于我们', '主办单位：', '网站标识码', '本网站支持IPv6', '政务新媒体矩阵', '旌德县人民政府'):
        i = body.find(kw)
        if i > 0:
            body = body[:i]
    return body.strip()


def parse_detail(html_text):
    title = ""
    m = re.search(r'name="ArticleTitle"\s+content="([^"]+)"', html_text)
    if not m:
        m = re.search(r'ArticleTitle"\s+content="([^"]+)"', html_text)
    if m:
        title = re.sub(r"\s+", " ", m.group(1)).strip()
    pub_date = ""
    m = re.search(r'name="PubDate"\s+content="([^"]+)"', html_text)
    if not m:
        m = re.search(r'PubDate"\s+content="([^"]+)"', html_text)
    if m:
        pub_date = m.group(1).strip()[:10]
    body = extract_zoom(html_text)
    if not body:
        return None
    body = clean_body(body)
    if not body:
        return None
    return {"title": title, "pub_date": pub_date, "body": body}


def html_to_text(content, page_url):
    """zoom 正文 → 存储格式: 附件绝对化 + 表格保留 + 段落化."""
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    content2 = content
    content2 = re.sub(r'<img[^>]*src="[^"]*fileTypeImages/icon_[a-z]+\.gif"[^>]*/?>', "", content2, flags=re.I)
    attachments = []
    link_protect = {}

    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = txt.strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key

    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    table_protect = []

    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"

    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)
    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key or key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()
    return out, has_table, attachments


async def run_async(pages=5):
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled", "--no-sandbox",
        ])
        ctx = await browser.new_context(
            user_agent=UA, ignore_https_errors=True,
            viewport={"width": 1366, "height": 900},
        )
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 1. 过盾
        ok = False
        for i in range(5):
            try:
                resp = await page.goto(list_url(1), wait_until="domcontentloaded", timeout=60000)
                await page.wait_for_timeout(2500)
                html_text = await page.content()
                if "OpennessContent/show" in html_text or "search-list" in html_text:
                    ok = True
                    break
                print(f"  [WAF] attempt={i} status={resp.status if resp else '?'} 无列表特征")
            except Exception as e:
                print(f"  [WAF] goto retry {i}: {str(e)[:60]}")
            await page.wait_for_timeout(2500)
        if not ok:
            print("  [FATAL] 过盾失败")
            await browser.close()
            return 0

        # 2. 列表
        items_all = []
        cutoff_hit = False
        total_pages = pages
        for pg in range(1, pages + 1):
            url = list_url(pg)
            try:
                html_text = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", url)
            except Exception as e:
                print(f"  [LIST-ERR] page={pg}: {e}")
                continue
            items = parse_list(html_text)
            print(f"  第{pg}页: {len(items)} 条")
            if not items:
                if pg > 1:
                    break
                continue
            m_pc = re.search(r'共(\d+)页', html_text)
            if m_pc:
                total_pages = max(total_pages, int(m_pc.group(1)))
            for it in items:
                if it["date"] and it["date"] < CUTOFF:
                    cutoff_hit = True
                    print(f"  [CUTOFF] 第{pg}页出现 < {CUTOFF} 日期, 停止翻页")
                    break
                items_all.append(it)
            if cutoff_hit:
                break
            await page.wait_for_timeout(random.randint(1500, 3000))
            if pg >= total_pages:
                break

        print(f"  [LIST] 共收集 {len(items_all)} 条, total_pages={total_pages}")
        if not items_all:
            await browser.close()
            return 0

        # 3. 详情
        records = []
        for i, it in enumerate(items_all, 1):
            url = it["url"]
            try:
                html_text = await page.evaluate("""async (url) => {
                    const r = await fetch(url, {credentials: 'include'});
                    return await r.text();
                }""", url)
            except Exception as e:
                print(f"  [{i}/{len(items_all)}] [ERR] {e}")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": it["date"],
                    "url": url, "content": f"<p>{it['title']}</p>",
                })
                continue
            d = parse_detail(html_text)
            if not d:
                print(f"  [{i}/{len(items_all)}] [NO-BODY] {it['title'][:40]}")
                records.append({
                    "site_name": SITE_NAME, "title": it["title"], "pub_date": it["date"],
                    "url": url, "content": f"<p>{it['title']}</p>",
                })
                continue
            title = d["title"] or it["title"]
            pub_date = d["pub_date"] or it["date"]
            content, has_table, attachments = html_to_text(d["body"], url)
            if len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                print(f"  [{i}/{len(items_all)}] [EMPTY] {title[:40]}")
                content = f"<p>{title}</p>"
            records.append({
                "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
                "url": url, "content": content, "attachments": json.dumps(attachments, ensure_ascii=False),
            })
            print(f"  [{i}/{len(items_all)}] 最新: {title[:60]}...")
            await page.wait_for_timeout(random.randint(1000, 2000))

        new_c = push_to_searchdb(records, batch_label="ahjd_hpgs")
        await browser.close()
        return new_c


if __name__ == "__main__":
    n = asyncio.run(run_async(pages=_PAGES))
    print(f"Done: {n} records")
