#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""弋阳县人民政府 (www.jxyy.gov.cn) 通知公告爬虫 — UCAP CMS + 电信NWAF + 瑞数双层防护
用法: python3 crawl_jxyy_tzgg.py [--pages N] [--out /tmp/jxyy_tzgg.jsonl]
过盾: 服务器 playwright + playwright_stealth.Stealth (NWAF 405 拉黑本地IP, 服务器IP可过)
列表: li > h4 > a[title] + span.time, 分页 list.shtml / list_{N}.shtml
详情: div#zoomcon > ucapcontent, meta ArticleTitle/PubDate/ContentSource
"""
import argparse, json, re, sys, time, random
from urllib.parse import urljoin
import os

BASE = 'http://www.jxyy.gov.cn'
LIST_URL = BASE + '/jxyy/tzgg/list.shtml'
SITE_NAME = 'jxyy_tzgg'
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36'

BLOCK_TITLES = ('请稍候', 'NWAF', '405', '访问', '拦截')


def clean_html(html, base_url=None):
    """白名单清洗: 保留 <p>/<table>/<a>, unwrap 装饰标签, 清 inline style, 链接绝对化
    base_url: 详情页 URL (图片/附件相对路径以此为基址)"""
    from bs4 import BeautifulSoup
    base = base_url or BASE
    soup = BeautifulSoup(html, 'html.parser')
    # video 先提取转链接 (内部含 object/embed, 需先取 src)
    for v in soup.find_all('video'):
        src = v.get('src') or ''
        if not src:
            s = v.find('source')
            if s:
                src = s.get('src', '')
        if not src:
            p = v.find('param', attrs={'name': 'url'})
            if p:
                src = p.get('value', '')
        if not src:
            e = v.find('embed')
            if e:
                src = e.get('src', '')
        if src:
            abs_src = src if src.startswith('http') else urljoin(base, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看视频'
            v.replace_with(a)
        else:
            v.decompose()
    for t in soup(['script', 'style', 'iframe', 'object', 'embed']):
        t.decompose()
    # 清 inline style / class / 无障碍属性
    for t in soup.find_all(True):
        for attr in ('style', 'class', 'align', 'valign', 'border', 'cellpadding', 'cellspacing',
                     'width', 'height', 'bgcolor', 'face', 'color', 'size', 'lang', 'dir',
                     'setedaria', 'tabindex', 'role'):
            if attr in t.attrs:
                del t[attr]
        for attr in list(t.attrs):
            if attr.startswith('aria-'):
                del t[attr]
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(base, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label', 'h1', 'h2', 'h3', 'ucapcontent']):
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(base, href)
        elif href:
            a.decompose()
    # 空段落清理 (保留含表格/链接的)
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('a'):
            p.decompose()
    return str(soup)


def parse_list_html(html):
    """列表页解析: li > h4 > a[title] + span.time"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for li in soup.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a.get('href', '')
        # 只保留详情页: /jxyy/tzgg/YYYYMM/{hash}.shtml
        if not re.search(r'/jxyy/tzgg/20\d{2}\d{2}/[a-f0-9]+\.shtml', href):
            continue
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        span = li.find('span', class_='time')
        date = span.get_text(strip=True) if span else ''
        if title and href.endswith('.shtml'):
            items.append({'title': title, 'url': urljoin(BASE, href), 'date': date})
    # 去重 (按 url)
    seen, out = set(), []
    for it in items:
        if it['url'] not in seen:
            seen.add(it['url'])
            out.append(it)
    return out


def parse_detail_html(html, url):
    """详情页解析: meta 字段 + #zoomcon > ucapcontent"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    def meta(name):
        m = soup.find('meta', attrs={'name': name})
        return (m.get('content') or '').strip() if m else ''
    title = meta('ArticleTitle') or ''
    pub = meta('PubDate') or ''
    source = meta('ContentSource') or ''
    zoom = soup.find('div', id='zoomcon')
    if not zoom:
        zoom = soup.find('div', class_='article-content-body')
    body = ''
    if zoom:
        ucap = zoom.find('ucapcontent')
        if not ucap:
            ucap = zoom
        # 分离附件链接与正文
        att_links = []
        for a in ucap.find_all('a', href=True):
            h = a['href'].lower()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj|wps|et|jpg|jpeg|png|gif)$', h):
                t = a.get_text(strip=True) or '附件'
                fu = h if h.startswith('http') else urljoin(url, h)
                att_links.append((t, fu))
                a.decompose()
        body = clean_html(str(ucap), url)
        if att_links:
            att_p = ''.join(f'<p><a href="{fu}">{t}</a></p>' for t, fu in att_links)
            body = (body + '\n' + att_p).strip()
    return {
        'title': title,
        'publish_date': pub[:10] if pub else '',
        'content': body,
        'source': source,
        'attachments': [{'fileName': t, 'fileUrl': fu} for t, fu in att_links] if body else [],
    }


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=0, help='0=自动探测全量')
    ap.add_argument('--out', default='/tmp/jxyy_tzgg.jsonl')
    ap.add_argument('--max-detail-fail', type=int, default=10)
    args = ap.parse_args()

    from playwright.sync_api import sync_playwright
    from playwright_stealth import Stealth

    stats = {'list_ok': 0, 'items': 0, 'detail_ok': 0, 'detail_fail': 0, 'empty': 0}
    out_fp = open(args.out, 'w', encoding='utf-8')
    try:
        with sync_playwright() as p:
            browser = p.chromium.launch(
                headless=True,
                args=["--disable-blink-features=AutomationControlled", "--no-sandbox", "--disable-dev-shm-usage"])
            ctx = browser.new_context(
                user_agent=UA, viewport={"width": 1366, "height": 900},
                locale="zh-CN", timezone_id="Asia/Shanghai")
            page = ctx.new_page()
            Stealth().apply_stealth_sync(page)

            def wait_ok(url, tries=15):
                """goto + 等待过盾 (title 不含屏蔽词)"""
                try:
                    page.goto(url, timeout=60000, wait_until='domcontentloaded')
                except Exception:
                    pass
                for i in range(tries):
                    time.sleep(3)
                    t = page.title() or ''
                    if not any(b in t for b in BLOCK_TITLES) and t.strip():
                        return True
                return False

            # 1. 首列表页过盾 + 探测总页数
            if not wait_ok(LIST_URL):
                print('过盾失败 (列表页)', flush=True)
                return
            html = page.content()
            pages_total = 1
            m = re.findall(r'list_(\d+)\.shtml', html)
            if m:
                pages_total = max(int(x) for x in m)
            if args.pages > 0:
                pages_total = min(pages_total, args.pages)
            print(f'过盾成功, 总页数={pages_total}, title={page.title()[:40]}', flush=True)

            # 2. 抓列表
            all_items = []
            for pg in range(1, pages_total + 1):
                url = LIST_URL if pg == 1 else LIST_URL.replace('list.shtml', f'list_{pg}.shtml')
                if pg > 1:
                    if not wait_ok(url):
                        print(f'页{pg} 过盾失败', flush=True)
                        break
                    html = page.content()
                items = parse_list_html(html)
                all_items.extend(items)
                print(f'页{pg}: {len(items)} 条, 累计 {len(all_items)}', flush=True)
                time.sleep(random.uniform(1.2, 2.0))
            stats['items'] = len(all_items)
            print(f'列表完成: {len(all_items)} 条', flush=True)

            # 3. 抓详情 (增量: DB 已有 page_url 跳过)
            existing = set()
            try:
                import sqlite3
                dbp = os.environ.get('SEARCH_DB', '/root/search.db')
                conn = sqlite3.connect(dbp)
                conn.execute("PRAGMA busy_timeout=10000")
                for row in conn.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
                    existing.add(row[0])
                conn.close()
            except Exception:
                pass
            print(f'DB 已有 {len(existing)} 条, 待抓 {len(all_items) - len(existing & set(i["url"] for i in all_items))} 条', flush=True)
            for idx, it in enumerate(all_items, 1):
                if it['url'] in existing:
                    continue
                ok = wait_ok(it['url'], tries=8)
                if not ok:
                    stats['detail_fail'] += 1
                    print(f'详情 {idx}/{len(all_items)}: 过盾失败 {it["url"]}', flush=True)
                    if stats['detail_fail'] >= args.max_detail_fail:
                        print('失败过多, 中止', flush=True)
                        break
                    continue
                d = parse_detail_html(page.content(), it['url'])
                if not d['title']:
                    d['title'] = it['title']
                if not d['publish_date']:
                    d['publish_date'] = it['date']
                content = d['content']
                if len(content) < 10:
                    stats['empty'] += 1
                rec = {
                    'title': d['title'],
                    'publish_date': d['publish_date'],
                    'content': content,
                    'page_url': it['url'],
                    'source_url': it['url'],
                    'site_name': SITE_NAME,
                    'summary': re.sub(r'<[^>]+>', ' ', content).strip()[:200],
                    'date_rank': 0,
                    'author': d['source'],
                    'content_source': d['source'],
                    'attachments': d['attachments'],
                }
                out_fp.write(json.dumps(rec, ensure_ascii=False) + '\n')
                stats['detail_ok'] += 1
                if idx % 10 == 0 or idx == len(all_items):
                    print(f'详情 {idx}/{len(all_items)}: ok={stats["detail_ok"]} fail={stats["detail_fail"]}', flush=True)
                time.sleep(random.uniform(0.8, 1.5))
            browser.close()
    finally:
        out_fp.close()
    print(f'完成: {json.dumps(stats, ensure_ascii=False)}', flush=True)


if __name__ == '__main__':
    main()
