#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
惠安县人民政府 - 通知公告 (tzgg) 爬虫
列表: https://www.huian.gov.cn/zwgk/xwzx/gggs/tzgg/
  avalon.js 前端渲染 (terton.dynamicAndStaticData), 静态 HTML 含 JS 数组(每页15条)
  翻页: 点页码按钮 (ms-controller=list_pagebar), recordCount=1421 约95页
详情: /zwgk/xwzx/gggs/tzgg/YYYYMM/tYYYYMMDD_ID.htm
  标题: 页面 <title> 或 h1
  正文: div.TRS_UEDITOR 或类似容器
用法: python3 crawl_huian_gsgg.py [--pages=N] [--limit=N]
"""
import os, sys, re, time, json, html as html_mod
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "https://www.huian.gov.cn"
LIST_URL = BASE_URL + "/zwgk/xwzx/gggs/tzgg/"
SITE_NAME = "惠安县人民政府-通知公告"
CATEGORY = "通知公告"
MAX_PAGES = 95   # recordCount=1421 / 15 per page

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"


def parse_args():
    pages, limit = 0, 0
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a.startswith("--limit="):
            limit = int(a.split("=")[1])
    return pages, limit


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def extract_items(page):
    """从渲染后页面提取 doctitle/docpuburl/docreltime"""
    items = page.evaluate("""() => {
        const out = [];
        // avalon 渲染后列表: 通常 ul li a
        const items = document.querySelectorAll('ul li a[href*=".htm"], .list_base a[href*=".htm"]');
        const seen = new Set();
        for (const a of items) {
            const href = a.getAttribute('href') || '';
            if (!href.includes('.htm') || seen.has(href)) continue;
            seen.add(href);
            let date = '';
            const li = a.closest('li');
            if (li) {
                const m = li.textContent.match(/(\\d{4})[-\\/](\\d{1,2})[-\\/](\\d{1,2})/);
                if (m) date = m[1] + '-' + m[2].padStart(2,'0') + '-' + m[3].padStart(2,'0');
            }
            const title = (a.textContent || '').trim();
            if (title.length >= 4) out.push({href: href, title: title, date: date});
        }
        return out;
    }""")
    # 统一转绝对 URL (处理 ./ 前缀)
    for it in items:
        href = it["href"]
        if href.startswith("./"):
            href = LIST_URL + href[2:]
        elif not href.startswith("http"):
            href = urljoin(LIST_URL, href)
        it["href"] = href
    return items


def click_page(page, pg):
    """点击页码按钮"""
    return page.evaluate("""(pg) => {
        // avalon pagebar 页码链接
        const links = document.querySelectorAll('.page_base a, .pagebar a, .pgStyle a, a[href="javascript:void(0)"], a[href="#"]');
        for (const a of links) {
            const t = (a.textContent || '').trim();
            if (t === String(pg)) { a.click(); return true; }
        }
        // 下一页
        const next = document.querySelector('.page_base .next a, .pagebar .next a, a[title="下一页"], .nextpage');
        if (next) { next.click(); return true; }
        return false;
    }""", pg)


def fetch_detail(page, url):
    """详情页: h1/标题 + 正文容器"""
    try:
        page.goto(url, wait_until="commit", timeout=30000)
        page.wait_for_timeout(1500)
        html = page.content()
        title = ""
        m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
        if m:
            title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<h1[^>]*>([^<]+)</h1>", html)
            if m:
                title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<title>([^<]+)</title>", html)
            if m:
                title = clean_title(m.group(1).split("_")[0])
        pub_date = ""
        m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
        if m:
            pub_date = m.group(1).strip()[:10]
        if not pub_date:
            m = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", html)
            if m:
                pub_date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        # 正文容器 (兼容 class="X" 和 class=X 两种写法)
        body = ""
        for cid in ['TRS_Editor', 'TRS_UEDITOR', 'zoom', 'content', 'article-content', 'zoomcon', 'wzcon', 'article_content_01']:
            # 匹配 class="cid" / class=cid / id="cid" / id=cid
            m = re.search(r'(?:class|id)\s*=\s*["\']?[^"\'>]*\b' + re.escape(cid) + r'\b[^"\'>]*["\']?', html)
            if not m:
                continue
            i = m.start()
            start = html.find(">", i) + 1
            depth = 1  # 已进入容器本身，从 1 开始
            j = start
            while j < len(html):
                if html[j:j+4] == "<div":
                    depth += 1
                    j += 4
                elif html[j:j+6] == "</div>":
                    depth -= 1
                    j += 6
                    if depth == 0:
                        break
                else:
                    j += 1
            raw = html[start:j-6]
            raw = re.sub(r"<script[\s\S]*?</script>", "", raw)
            raw = re.sub(r"</?body[^>]*>", "", raw)
            raw = re.sub(r"<style[\s\S]*?</style>", "", raw)
            raw = re.sub(r'href="([^"]*)"', lambda m: 'href="%s"' % (urljoin(url, m.group(1)) if not m.group(1).startswith(("http", "#", "javascript")) else m.group(1)), raw)
            raw = re.sub(r'src="([^"]*)"', lambda m: 'src="%s"' % (urljoin(url, m.group(1)) if not m.group(1).startswith(("http", "data:", "javascript")) else m.group(1)), raw)
            raw = re.sub(r'\sstyle="[^"]*"', "", raw)
            raw = re.sub(r"<span[^>]*>|</span>|<strong[^>]*>|</strong>|<b[^>]*>|</b>|<font[^>]*>|</font>", "", raw)
            body = raw.strip()
            if len(body) >= 20:
                break
        if not body or len(body) < 20:
            body = ""
        return title, pub_date, body
    except Exception as e:
        print(f"    [ERR] detail {url}: {e}", flush=True)
        return "", "", ""


def main():
    max_pages, limit = parse_args()
    if not max_pages:
        max_pages = MAX_PAGES
    print(f"[Huian] SITE={SITE_NAME} max_pages={max_pages}", flush=True)

    from playwright.sync_api import sync_playwright
    all_items = []
    seen = set()

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--disable-blink-features=AutomationControlled", "--no-sandbox"])
        ctx = browser.new_context(user_agent=UA, locale="zh-CN")
        page = ctx.new_page()
        page.route("**/*", lambda route: route.abort() if route.request.resource_type in ("image", "media", "font", "stylesheet") else route.continue_())

        page.goto(LIST_URL, wait_until="commit", timeout=30000)
        page.wait_for_timeout(4000)

        for pg in range(1, max_pages + 1):
            if pg > 1:
                clicked = False
                for attempt in range(3):
                    try:
                        clicked = click_page(page, pg)
                        if clicked:
                            break
                        page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
                        page.wait_for_timeout(800)
                    except Exception as e:
                        print(f"    [WARN] page {pg} click err {e}", flush=True)
                        page.wait_for_timeout(1000)
                if not clicked:
                    print(f"  [STOP] page {pg} no click target", flush=True)
                    break
                page.wait_for_timeout(2000)

            items = extract_items(page)
            new_count = 0
            for it in items:
                href = it["href"]
                if href.startswith("./"):
                    href = LIST_URL + href[2:]
                elif not href.startswith("http"):
                    href = urljoin(BASE_URL, href)
                if href in seen:
                    continue
                seen.add(href)
                all_items.append(it)
                new_count += 1
            print(f"  Page {pg}: {len(items)}条(新{new_count}) 累计{len(all_items)}", flush=True)

            # 是否到尾页
            has_next = page.evaluate("""() => {
                const n = document.querySelector('.page_base .next a, .pagebar .next a, a[title="下一页"]');
                if (!n) return false;
                return !(n.className.includes('disabled') || n.style.display === 'none');
            }""")
            if not has_next and pg > 1:
                print(f"  [END] page {pg}", flush=True)
                break
            if len(all_items) >= 1421:
                print(f"  [CUTOFF] reached {len(all_items)}", flush=True)
                break

        browser.close()

    print(f"列表完成: {len(all_items)} 条, 抓详情...", flush=True)
    if limit and len(all_items) > limit:
        all_items = all_items[:limit]

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--disable-blink-features=AutomationControlled", "--no-sandbox"])
        ctx = browser.new_context(user_agent=UA, locale="zh-CN")
        page = ctx.new_page()
        page.route("**/*", lambda route: route.abort() if route.request.resource_type in ("image", "media", "font", "stylesheet") else route.continue_())
        results = []
        for i, it in enumerate(all_items):
            d_title, d_date, body = fetch_detail(page, it["href"])
            title = d_title or it["title"]
            pub_date = d_date or it.get("date", "")
            summary = re.sub(r"<[^>]+>", " ", body) if body else title
            summary = re.sub(r"\s+", " ", summary).strip()[:300]
            atts = []
            if body:
                for m in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd|wps))"[^>]*>([^<]*)</a>', body, re.I):
                    atts.append({"name": m.group(2).strip(), "url": m.group(1)})
            results.append({
                "site_name": SITE_NAME, "source_url": it["href"], "url": it["href"],
                "title": title, "pub_date": pub_date, "content": body,
                "summary": summary, "category": CATEGORY,
                "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
            })
            if (i + 1) % 10 == 0:
                print(f"  detail {i+1}/{len(all_items)}", flush=True)
            time.sleep(0.3)
        browser.close()

    print(f"  pushing {len(results)} items to searchdb...", flush=True)
    push_to_searchdb(results, batch_label=SITE_NAME)
    empty = sum(1 for r in results if not (r.get("content") or "").strip())
    print(f"  DONE. pushed={len(results)} 空正文={empty}", flush=True)


if __name__ == "__main__":
    main()
