#!/usr/bin/env python3
"""
crawl_jxxd.py — 萍乡市湘东区人民政府（jxxd.gov.cn）公共企事业单位信息 爬虫
================================================================================
站点: 万维网(wm) 浏览器隔离防爬系统（SockJS + DOM命令推送 + about:blank链接）
必须用 Playwright 真实浏览器渲染：
  - 列表页: 渲染后读取 #tableList 表格（标题+日期），真实鼠标点击分页翻页
  - 详情页: 真实鼠标点击列表项 → 跳转 content/<id>/content_<id>.html → 提取正文
  - JS 合成 click() 会被 wm 过滤，必须用 Playwright locator.click()（真实事件）

栏目: c100253 公共企事业单位信息（421条/29页）
"""
import sys, os, re, time, json, sqlite3, html as html_lib, random
from datetime import datetime

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
try:
    from crawler_lib import push_to_searchdb, classify_industry
except Exception:
    push_to_searchdb = None

BASE_URL = "https://www.jxxd.gov.cn"
LIST_URL = "https://www.jxxd.gov.cn/jxxd/c100253/pc/list.html"
SITE_NAME = "萍乡市湘东区-公共企事业单位信息"
GROUP_NAME = "江西"
SCRIPT_NAME = "crawl_jxxd.py"
DB_PATH = os.environ.get("SEARCH_DB", "/root/search.db")

MAX_PAGES = 1   # 日跑默认只抓第1页；全量 --pages=29

def parse_args():
    global MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a.startswith("--pages="):
            try: MAX_PAGES = int(a.split("=", 1)[1])
            except ValueError: pass
        elif a == "--pages" and i + 1 < len(sys.argv):
            try: MAX_PAGES = int(sys.argv[i + 1])
            except ValueError: pass

def extract_list_items(page):
    """从渲染后的列表页提取 (标题, 发布日期)"""
    items = page.evaluate("""
        () => {
            const rows = document.querySelectorAll('#tableList table tr');
            const out = [];
            for (const tr of rows) {
                const a = tr.querySelector('a[target=_top]');
                const dateTd = tr.querySelector('td.fbrq');
                if (a) {
                    out.push({
                        title: a.innerText.trim(),
                        date: dateTd ? dateTd.innerText.trim() : ''
                    });
                }
            }
            return out;
        }
    """)
    return [(it["title"], it["date"]) for it in items if it["title"]]

_last_first_title = {}  # 每页第一条标题（翻页成功判断用）

def click_page(page, page_num):
    """点击分页按钮翻到指定页（重试3次）"""
    for attempt in range(3):
        try:
            # 找分页数字链接
            clicked = page.evaluate("""
                (target) => {
                    const links = [...document.querySelectorAll('a')];
                    const btn = links.find(a => a.innerText.trim() === String(target) && a.getAttribute('href') === 'javascript:;');
                    if (btn) { btn.scrollIntoView(); return true; }
                    return false;
                }
            """, page_num)
            if not clicked:
                print(f"  未找到第{page_num}页按钮")
                return False
            page.wait_for_timeout(1500 + random.randint(0, 1000))
            loc = page.locator("a[href='javascript:;']", has_text=str(page_num)).first
            loc.click()
            # 等待列表更新（第一条标题变化）
            deadline = time.time() + 30
            ok = False
            while time.time() < deadline:
                first = page.evaluate("""() => {
                    const a = document.querySelector('#tableList table a[target=_top]');
                    return a ? a.innerText.trim() : '';
                }""")
                if first and first != _last_first_title.get(page_num - 1, ''):
                    ok = True
                    break
                page.wait_for_timeout(2000)
            if ok:
                page.wait_for_timeout(4000)  # 等 SockJS 推送稳定
                return True
            print(f"  翻到第{page_num}页第{attempt+1}次失败（列表未更新）")
            page.wait_for_timeout(5000)
        except Exception as e:
            print(f"  翻页 {page_num} 第{attempt+1}次异常: {str(e)[:80]}")
            page.wait_for_timeout(5000)
    return False

def fetch_detail(page, ctx, idx):
    """Ctrl+click 列表第 idx 条在新tab打开详情 → 返回 (标题, 日期, 正文HTML, 附件, 详情URL)
    列表页状态保持不动，详情 tab 抓完即关。失败重试3次"""
    detail_url = ""
    for attempt in range(3):
        try:
            loc = page.locator('#tableList table a[target=_top]').nth(idx)
            loc.scroll_into_view_if_needed()
            page.wait_for_timeout(800 + random.randint(0, 500))
            with page.expect_popup(timeout=15000) as popup_info:
                loc.click(modifiers=["Control"])
            pop = popup_info.value
            pop.wait_for_load_state("domcontentloaded", timeout=25000)
            # 等正文渲染
            try:
                pop.wait_for_selector("h1.article-title", timeout=20000)
            except Exception:
                pass
            pop.wait_for_timeout(4000)
            detail_url = pop.url
            html_text = pop.content()
            pop.close()
            break
        except Exception as e:
            print(f"  详情 {idx} 第{attempt+1}次失败: {str(e)[:100]}")
            # 关闭可能残留的新tab
            for pg in ctx.pages:
                if pg != page:
                    try: pg.close()
                    except Exception: pass
            page.wait_for_timeout(4000)
            continue
    if not detail_url:
        return "", "", "", [], ""
    # 提取
    title = ""
    date = ""
    content_html = ""
    attachments = []
    try:
        m = re.search(r'<h1[^>]*class="article-title"[^>]*>(.*?)</h1>', html_text, re.S)
        if m:
            title = clean_title(m.group(1))
    except Exception:
        pass
    m = re.search(r'id="data-pubdate"[^>]*>([^<]+)<', html_text)
    if m:
        date = m.group(1).strip()[:10]
    m = re.search(r'<div class="article-content" id="zoomcon">(.*?)</div>\s*<div class="article-appendix"', html_text, re.S)
    if m:
        content_html = m.group(1)
    else:
        m = re.search(r'<div class="article-content" id="zoomcon">(.*?)</div>', html_text, re.S)
        if m:
            content_html = m.group(1)
    # 附件区
    appendix = re.search(r'<div class="article-appendix">(.*?)</div>\s*<div class="com-title square">', html_text, re.S)
    if appendix:
        for m3 in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*>([^<]{1,80})</a>', appendix.group(1), re.I):
            href = m3.group(1)
            if href.startswith("about:blank") or href.startswith("javascript"):
                continue
            if not href.startswith("http"):
                href = BASE_URL + href
            attachments.append((m3.group(2).strip(), href))
    return title, date, content_html, attachments, detail_url

def clean_title(t):
    t = html_lib.unescape(t)
    t = re.sub(r"<[^>]+>", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()

def store_record(conn, url, title, content, date, has_table, attachments):
    if not url or not title:
        return False
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, SITE_NAME))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table, attachments)
        raise

def main():
    parse_args()
    from playwright.sync_api import sync_playwright
    conn = sqlite3.connect(DB_PATH, timeout=120)
    new_count = 0
    skip_count = 0
    fail_count = 0
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=[
            "--no-sandbox", "--disable-blink-features=AutomationControlled"])
        ctx = browser.new_context(
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
            viewport={"width": 1366, "height": 768},
            locale="zh-CN", timezone_id="Asia/Shanghai")
        page = ctx.new_page()
        page.add_init_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})")
        page.goto(LIST_URL, timeout=60000, wait_until="domcontentloaded")
        # 等列表渲染（wm SockJS 推送可能较慢，最长等 25s）
        try:
            page.wait_for_selector("#tableList table a[target=_top]", timeout=25000)
        except Exception:
            pass
        page.wait_for_timeout(3000)
        # 翻页遍历
        for pg in range(1, MAX_PAGES + 1):
            if pg > 1:
                if not click_page(page, pg):
                    print(f"翻到第{pg}页失败，停止")
                    break
                page.wait_for_timeout(3000)
            items = extract_list_items(page)
            _last_first_title[pg] = items[0][0] if items else ""
            print(f"第{pg}页: {len(items)} 条")
            for idx, (title, date) in enumerate(items):
                # 列表级预查重：标题+日期已存在 → 跳过，不开详情（日跑提速关键）
                try:
                    dup = conn.execute(
                        "SELECT 1 FROM gov_raw WHERE site_name=? AND title=? AND publish_date=?",
                        (SITE_NAME, title, date),
                    ).fetchone()
                    if dup:
                        skip_count += 1
                        continue
                except Exception:
                    pass
                d_title, d_date, content_html, attachments, d_url = fetch_detail(page, ctx, idx)
                if not d_url:
                    fail_count += 1
                    print(f"  ⚠️ 详情失败跳过: {title[:50]}")
                    time.sleep(2)
                    continue
                if not d_title:
                    d_title = title
                if not d_date:
                    d_date = date
                has_table = 1 if "<table" in (content_html or "") else 0
                # 正文保留 HTML（表格/内嵌链接勿拍平），清理脚本与样式
                body = content_html or ""
                body = re.sub(r"<script.*?</script>", "", body, flags=re.S)
                body = re.sub(r"<style.*?</style>", "", body, flags=re.S)
                # 图片/链接相对路径转绝对
                body = re.sub(r'(<img[^>]*src=")([^"]+)(")', lambda m: m.group(1) + (m.group(2) if m.group(2).startswith("http") else BASE_URL + m.group(2)) + m.group(3), body)
                body = re.sub(r'(<a[^>]*href=")([^"]+)(")', lambda m: m.group(1) + (m.group(2) if m.group(2).startswith("http") else BASE_URL + m.group(2)) + m.group(3), body)
                body = body.strip()
                # 附件区 → 名称内嵌URL段落（不保留div/ul/li包装，多附件独立成段）
                if attachments:
                    att_parts = []
                    for name, url in attachments:
                        name = re.sub(r"<[^>]+>", "", name).strip()
                        att_parts.append(f'<p><a href="{url}">{name}</a></p>')
                    body += "\n" + "\n".join(att_parts)
                if store_record(conn, d_url, d_title, body[:500000], d_date, has_table, attachments):
                    new_count += 1
                    print(f"  ✅ {d_title[:50]} ({d_date})")
                else:
                    skip_count += 1
                conn.commit()  # 逐条提交，中途崩溃不丢已入库数据
                time.sleep(1 + random.random())
        browser.close()
    conn.commit()
    conn.close()
    print(f"\n完成! 新增: {new_count} 条, 跳过: {skip_count} 条, 点击失败: {fail_count} 条")

if __name__ == "__main__":
    main()
