#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""cnjg.gov.cn (剑阁县人民政府-公示公告) 爬虫 — 雷池 WAF 过盾+抓取一体
用法: python3 crawl_cnjg_local.py [--pages N] [--out /tmp/cnjg.jsonl]
过盾: 启动后 Chrome 窗口弹出, 请手动点击"确认"(脚本也会自动点, 双保险)
注意: 雷池 cookie 绑定浏览器指纹, 必须在同一 context 内抓全部页面
"""
import sys, time, re, json, argparse, random
sys.stdout.reconfigure(encoding='utf-8')
from playwright.sync_api import sync_playwright

BASE = "http://www.cnjg.gov.cn"
LIST_URL = BASE + "/new/list/20201010141948714.html"  # 公示公告
HOST_IP = "61.188.214.49"
UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
PAGE_SIZE = 20

STEALTH_JS = """
Object.defineProperty(navigator, 'webdriver', {get: () => undefined});
window.chrome = window.chrome || {runtime: {}};
Object.defineProperty(navigator, 'languages', {get: () => ['zh-CN', 'zh', 'en']});
Object.defineProperty(navigator, 'plugins', {get: () => [1, 2, 3, 4, 5]});
"""

def clean_html(html):
    """白名单清洗: 保留 p/table/a, 去 inline style, unwrap span, 链接绝对化"""
    from bs4 import BeautifulSoup
    from urllib.parse import urljoin
    soup = BeautifulSoup(html, 'html.parser')
    for t in soup(['script', 'style', 'iframe', 'object']):
        t.decompose()
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(BASE, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label']):
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(BASE, href)
        elif href and href.startswith('javascript'):
            a.decompose()
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('img'):
            p.decompose()
    # 清空 title/来源 残留
    return str(soup)

def parse_list(html):
    """解析列表页 → [(title, url, date)]"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    box = soup.find('div', class_='news-list2')
    if not box:
        return items
    for li in box.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        url = a['href']
        if not url.startswith('http'):
            url = BASE + url
        sp = li.find('span')
        date = sp.get_text(strip=True) if sp else ''
        items.append({'title': title, 'url': url, 'date': date})
    return items

def parse_detail(html):
    """解析详情页 → (title, publish_date, body_html)"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    h1 = soup.find('h1')
    title = h1.get_text(strip=True) if h1 else ''
    # 日期: 找 2024-01-01 模式
    m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    pub = m.group(1) if m else ''
    box = soup.find('div', class_='msg-content')
    if box:
        body = clean_html(str(box))
    else:
        body = ''
    return title, pub, body

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=5)
    ap.add_argument('--out', default='/tmp/cnjg.jsonl')
    ap.add_argument('--host', default=HOST_IP)
    args = ap.parse_args()

    out_fp = open(args.out, 'w', encoding='utf-8')
    stats = {'list_ok': 0, 'list_fail': 0, 'detail_ok': 0, 'detail_fail': 0}
    all_items = []
    try:
        with sync_playwright() as p:
            browser = p.chromium.launch(
                channel="chrome", headless=False,
                args=["--disable-blink-features=AutomationControlled",
                      f"--host-resolver-rules=MAP www.cnjg.gov.cn {args.host}, MAP cnjg.gov.cn {args.host}"])
            ctx = browser.new_context(user_agent=UA, viewport={"width": 1366, "height": 768},
                                      locale="zh-CN", ignore_https_errors=True)
            ctx.add_init_script(STEALTH_JS)
            page = ctx.new_page()
            try:
                page.goto(LIST_URL, timeout=45000, wait_until="domcontentloaded")
            except Exception as e:
                print(f"goto 超时: {str(e)[:50]}", flush=True)
            print("🖥️ Chrome 窗口已打开 — 请手动点击'确认'完成雷池验证（脚本也会自动尝试）...", flush=True)
            ok = False
            for attempt in range(60):
                try:
                    btn = page.query_selector("#sl-check")
                    if btn:
                        try:
                            btn.click(timeout=2000)
                        except Exception:
                            pass
                except Exception:
                    pass
                time.sleep(3)
                try:
                    html = page.content()
                    is_ch = "challenge" in html.lower() or "safeline" in html.lower() or "客户端异常" in html
                    if not is_ch and len(html) > 5000 and 'news-list2' in html:
                        ok = True
                        break
                    if attempt % 5 == 4:
                        print(f"  等待中 {(attempt+1)*3}s ch={is_ch}", flush=True)
                except Exception:
                    pass
            if not ok:
                print("❌ 未过盾", flush=True)
                browser.close()
                sys.exit(1)
            print("✅ 过盾成功!", flush=True)

            # 抓列表页
            for pg in range(1, args.pages + 1):
                url = f"{LIST_URL}?page={pg}"
                try:
                    page.goto(url, timeout=30000, wait_until="domcontentloaded")
                except Exception:
                    pass
                time.sleep(1.5)
                html = page.content()
                items = parse_list(html)
                if not items:
                    print(f"页{pg}: 空(可能到底或解析失败) len={len(html)}", flush=True)
                    stats['list_fail'] += 1
                    break
                for it in items:
                    if it['url'] not in [x['url'] for x in all_items]:
                        all_items.append(it)
                print(f"页{pg}: {len(items)} 条, 累计 {len(all_items)}", flush=True)
                stats['list_ok'] += 1
                time.sleep(random.uniform(0.8, 1.5))

            print(f"共 {len(all_items)} 个详情链接", flush=True)
            # 抓详情页
            for i, it in enumerate(all_items):
                try:
                    page.goto(it['url'], timeout=30000, wait_until="domcontentloaded")
                except Exception:
                    pass
                time.sleep(1.2)
                html = page.content()
                title, pub, body = parse_detail(html)
                if title and body:
                    item = {
                        'title': title,
                        'url': it['url'],
                        'page_url': it['url'],
                        'publish_date': pub or it['date'],
                        'content': body,
                        'author': '',
                        'site_name': 'cnjg_gsgg',
                        'attachments': [],
                    }
                    out_fp.write(json.dumps(item, ensure_ascii=False) + '\n')
                    out_fp.flush()
                    stats['detail_ok'] += 1
                else:
                    stats['detail_fail'] += 1
                if (i + 1) % 10 == 0:
                    print(f"详情 {i+1}/{len(all_items)}: ok={stats['detail_ok']} fail={stats['detail_fail']}", flush=True)
                time.sleep(random.uniform(0.8, 1.4))
            browser.close()
    finally:
        out_fp.close()
    print(f"完成: 列表{stats['list_ok']} 详情ok={stats['detail_ok']} fail={stats['detail_fail']}", flush=True)

if __name__ == '__main__':
    main()
