#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
惠安县人民政府 - 公示信息 (gsxx) 爬虫
列表: https://www.huian.gov.cn/zwgk/xwzx/gggs/gsxx/
  avalon.js 前端渲染 (terton.dynamicAndStaticData), 静态 HTML 含 JS 数组(每页15条)
  翻页: 点页码按钮 (ms-controller=list_pagebar)
详情: /zwgk/xwzx/gggs/gsxx/YYYYMM/tYYYYMMDD_ID.htm
  标题: meta ArticleTitle / h1
  正文: TRS_Editor/TRS_UEDITOR 等容器 (平衡 div)
用法: python3 crawl_huian_gsxx.py [--pages=N]
"""
import os, sys, re, time, json, html as html_mod
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "https://www.huian.gov.cn"
LIST_URL = BASE_URL + "/zwgk/xwzx/gggs/gsxx/"
SITE_NAME = "惠安县人民政府-公示信息"
CATEGORY = "公示信息"
MAX_PAGES = 30
CUTOFF = "2023-08-10"

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"


def parse_args():
    pages = 0
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a.isdigit():
            pages = int(a)
    return pages


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def extract_json_items(html):
    """从内嵌 dynamicAndStaticData JSON 提取列表 (无 JS 执行)"""
    items = []
    # 匹配 terton.dynamicAndStaticData({ list: [...] })
    m = re.search(r"dynamicAndStaticData\(\s*\{[^}]*?list:\s*\[(.*?)\]\s*\)", html, re.S)
    if not m:
        m = re.search(r"list:\s*\[(.*?)\]\s*\}", html, re.S)
    if not m:
        return items
    blob = m.group(1)
    # 提取每个 { "docreltime": '...', "doctitle": '...', "docpuburl": '...' }
    for rec in re.finditer(r"\{\s*\"docreltime\"\s*:\s*'([^']*)'\s*,\s*\"doctitle\"\s*:\s*'([^']*)'\s*,\s*\"docpuburl\"\s*:\s*'([^']*)'\s*\}", blob):
        date, title, url = rec.group(1), clean_title(rec.group(2)), rec.group(3)
        if len(title) < 4:
            continue
        if url.startswith("./"):
            url = urljoin(LIST_URL, url[2:])
        elif url.startswith("/"):
            url = BASE_URL + url
        else:
            url = urljoin(LIST_URL, url)
        if not url.startswith(BASE_URL):
            continue
        items.append({"title": title, "href": url, "date": date})
    return items


def click_page(page, pg):
    """点击页码按钮 (数字按钮, 勿点 下一页 span)"""
    try:
        clicked = page.evaluate("""(t) => {
            const links = document.querySelectorAll('a[href="#"], .page_base a, .pagination a');
            for (const a of links) {
                if (a.textContent.trim() === String(t)) { a.click(); return true; }
            }
            return false;
        }""", pg)
        return clicked
    except Exception:
        return False


def extract_items(page):
    """从渲染后页面提取列表 (avalon 渲染完成)"""
    items = page.evaluate("""() => {
        const out = [];
        const seen = new Set();
        const anchors = document.querySelectorAll('a[href*=".htm"]');
        for (const a of anchors) {
            const href = a.getAttribute('href') || '';
            if (!href.includes('.htm') || seen.has(href)) continue;
            const title = (a.textContent || '').trim();
            if (title.length < 4) continue;
            seen.add(href);
            let date = '';
            const li = a.closest('li');
            const scope = li || a.parentElement;
            if (scope) {
                const m = scope.textContent.match(/(\\d{4})[-\\/](\\d{1,2})[-\\/](\\d{1,2})/);
                if (m) date = m[1] + '-' + m[2].padStart(2,'0') + '-' + m[3].padStart(2,'0');
            }
            out.push({href: href, title: title, date: date});
        }
        return out;
    }""")
    for it in items:
        href = it["href"]
        if href.startswith("./"):
            href = urljoin(LIST_URL, href[2:])
        elif href.startswith("/"):
            href = BASE_URL + href
        else:
            href = urljoin(LIST_URL, href)
        it["href"] = href
    return items


def fetch_detail(page, url):
    """详情页: 标题 + 正文容器 (平衡 div)"""
    try:
        page.goto(url, wait_until="commit", timeout=30000)
        page.wait_for_timeout(1200)
        html = page.content()
        title = ""
        m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
        if m:
            title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<h1[^>]*>([^<]+)</h1>", html)
            if m:
                title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<title>([^<]+)</title>", html)
            if m:
                title = clean_title(m.group(1).split("_")[0])
        pub_date = ""
        m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
        if m:
            pub_date = m.group(1).strip()[:10]
        if not pub_date:
            m = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", html)
            if m:
                pub_date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        body = ""
        for cid in ['TRS_Editor', 'TRS_UEDITOR', 'zoom', 'content', 'article-content', 'zoomcon', 'wzcon', 'article_content_01']:
            m = re.search(r'(?:class|id)\s*=\s*["\']?[^"\'>]*\b' + re.escape(cid) + r'\b[^"\'>]*["\']?', html)
            if not m:
                continue
            i = m.start()
            start = html.find(">", i) + 1
            depth = 1
            j = start
            while j < len(html):
                if html[j:j+4] == "<div":
                    depth += 1
                    j += 4
                elif html[j:j+6] == "</div>":
                    depth -= 1
                    j += 6
                    if depth == 0:
                        break
                else:
                    j += 1
            raw = html[start:j-6]
            raw = re.sub(r"<script[\s\S]*?</script>", "", raw)
            raw = re.sub(r"<style[\s\S]*?</style>", "", raw)
            # href/src 绝对化
            raw = re.sub(r'href="([^"]*)"', lambda m2: 'href="%s"' % (urljoin(url, m2.group(1)) if not m2.group(1).startswith(("http", "#", "javascript")) else m2.group(1)), raw)
            raw = re.sub(r'src="([^"]*)"', lambda m2: 'src="%s"' % (urljoin(url, m2.group(1)) if not m2.group(1).startswith(("http", "data:", "javascript")) else m2.group(1)), raw)
            raw = re.sub(r'\sstyle="[^"]*"', "", raw)
            raw = re.sub(r"<span[^>]*>|</span>|<strong[^>]*>|</strong>|<b[^>]*>|</b>|<font[^>]*>|</font>", "", raw)
            raw = re.sub(r'^\s*<div[^>]*>\s*', '', raw)
            # 规范化 <p> 标签
            raw = re.sub(r'<p[^>]*>', '<p>', raw)
            # 删除空段落
            raw = re.sub(r'<p>\s*&nbsp;\s*</p>', '', raw)
            raw = re.sub(r'<p>\s*</p>', '', raw)
            # 裸文本（无 <p> 包裹）→ <p> 包裹，确保 search_app HTML 渲染
            if '<p>' not in raw and '<table' not in raw:
                raw = '<p>' + raw + '</p>'
            body = raw.strip()
            if len(body) >= 20:
                break
        if not body or len(body) < 20:
            body = ""
        return title, pub_date, body
    except Exception as e:
        print(f"    [ERR] detail {url}: {e}", flush=True)
        return "", "", ""


def main():
    max_pages = parse_args() or MAX_PAGES
    print(f"[HUIAN-GSXX] SITE={SITE_NAME} max_pages={max_pages}", flush=True)

    from playwright.sync_api import sync_playwright
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--disable-blink-features=AutomationControlled", "--no-sandbox"])
        ctx = browser.new_context(user_agent=UA, locale="zh-CN")
        page = ctx.new_page()
        page.goto(LIST_URL, wait_until="commit", timeout=30000)
        page.wait_for_timeout(2500)

        all_items = []
        seen = set()
        for pg in range(1, max_pages + 1):
            items = extract_items(page)
            if not items:
                print(f"  Page {pg}: 0 items (end)", flush=True)
                break
            new_count = 0
            for it in items:
                if it["href"] in seen:
                    continue
                seen.add(it["href"])
                all_items.append(it)
                new_count += 1
            print(f"  Page {pg}: {len(items)}条(新{new_count}) 累计{len(all_items)}", flush=True)
            # 检查 CUTOFF
            min_date = min((it["date"] for it in items if it["date"]), default="")
            if min_date and min_date < CUTOFF:
                print(f"  [CUTOFF] page {pg} min_date={min_date} < {CUTOFF}, stop", flush=True)
                break
            if pg < max_pages:
                if not click_page(page, pg + 1):
                    print(f"  [END] page {pg+1} click fail", flush=True)
                    break
                page.wait_for_timeout(1800)

        print(f"列表完成: {len(all_items)} 条, 抓详情...", flush=True)
        results = []
        for i, it in enumerate(all_items):
            d_title, d_date, body = fetch_detail(page, it["href"])
            title = d_title or it["title"]
            pub_date = d_date or it.get("date", "")
            summary = re.sub(r"<[^>]+>", " ", body) if body else title
            summary = re.sub(r"\s+", " ", summary).strip()[:300]
            atts = []
            if body:
                for m in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd|wps))"[^>]*>([^<]*)</a>', body, re.I):
                    atts.append({"name": m.group(2).strip(), "url": m.group(1)})
            results.append({
                "site_name": SITE_NAME, "source_url": it["href"], "url": it["href"],
                "title": title, "pub_date": pub_date, "content": body,
                "summary": summary, "category": CATEGORY,
                "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
            })
            if (i + 1) % 10 == 0:
                print(f"  detail {i+1}/{len(all_items)}", flush=True)
            time.sleep(0.3)

        print(f"  pushing {len(results)} items...", flush=True)
        push_to_searchdb(results, batch_label=SITE_NAME)
        empty = sum(1 for r in results if not (r.get("content") or "").strip())
        print(f"  DONE. pushed={len(results)} 空正文={empty}", flush=True)
        browser.close()


if __name__ == "__main__":
    main()
