#!/usr/bin/env python3
"""crawl_tszwgktzgg.py - 唐山市人民政府-公告公示 (瑞数4代WAF, playwright无头浏览器过校验)

站点: https://www.tangshan.gov.cn/zhuzhan/tszwgktzgg/index.html
CMS: 天翼云WAF(Environment Checking JS challenge) + 服务端渲染页面
列表: /zhuzhan/tszwgktzgg/YYYYMMDD/ID.html 链接, index_N.html 分页 (实测7页, P8越界空)
详情: meta[ArticleTitle] + meta[PubDate] + div#conN 正文(15+个<p>段落, 文本在<label>内, 图片型公告保留<p><img>)
WAF: requests/curl 全部412 → playwright chromium headless + --disable-blink-features=AutomationControlled 可过
样本: 唐山市人民政府关于实施2026年度防空警报试鸣的通告 (2026-07-02)
"""
import sys, os, re, argparse, time
from datetime import datetime, timedelta
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=3 * 365)).strftime("%Y-%m-%d")
SITE_NAME = '唐山市人民政府-公告公示'
GROUP_NAME = '河北'
BASE = 'https://www.tangshan.gov.cn'
LIST_PATH = '/zhuzhan/tszwgktzgg/'
LIST_URL = BASE + LIST_PATH + 'index.html'
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'


def clean_body(html, detail_url):
    """清洗 div#conN 正文: 保留 <p> 段落 + 图片 + 段内附件<a>, 剥 label/span 内层"""
    m = re.search(r'<div id="conN">(.*?)</div>\s*</div>', html, re.DOTALL)
    if not m:
        m = re.search(r'<div id="conN">(.*)', html, re.DOTALL)
    if not m:
        return ''
    body = m.group(1)
    # 若 conN 后紧跟 </div> 被非贪婪吞掉, 重新平衡: 直接截到 </div> 前
    if body.count('<div') > body.count('</div>'):
        # 平衡到最后一个 </div> 前
        closes = [x.start() for x in re.finditer(r'</div>', body)]
        if closes:
            body = body[:closes[-1]]
    # 提取段落
    paras = re.findall(r'<p[^>]*>(.*?)</p>', body, re.DOTALL)
    out = []
    for p in paras:
        # 图片段落: 保留 <p><img src=绝对URL>
        if '<img' in p:
            for s in re.findall(r'<img[^>]+src="([^"]+)"[^>]*>', p):
                out.append(f'<p><img src="{urljoin(detail_url, s.strip())}"></p>')
            continue
        # 附件段落: 保留 <a> 链接 (绝对化), 剥其他标签
        p2 = re.sub(r'(<a[^>]+href="[^"]+"[^>]*>)', lambda mm: mm.group(1).replace('href="', 'href="ABSURL').replace('ABSURL', 'X'), p)
        # 绝对化 a href
        def _abs_a(mm):
            href = mm.group(1)
            return f'<a href="{urljoin(detail_url, href)}">'
        p2 = re.sub(r'<a[^>]+href="([^"]+)"[^>]*>', lambda mm: '<a href="' + urljoin(detail_url, mm.group(1)) + '">', p2)
        # 剥内层标签 (label/span/strong...), 保留 <a>
        parts = re.split(r'(<a[^>]*>.*?</a>)', p2, flags=re.DOTALL)
        cleaned_parts = []
        for part in parts:
            if part.startswith('<a '):
                cleaned_parts.append(part)
            else:
                txt = re.sub(r'<[^>]+>', '', part)
                txt = txt.replace('\u3000', ' ').replace('\xa0', ' ').replace('\u200b', '')
                txt = re.sub(r'\s+', ' ', txt).strip()
                if txt:
                    cleaned_parts.append(txt)
        final = ''.join(cleaned_parts).strip()
        if final:
            out.append(f'<p>{final}</p>')
    # 无 <p> 段落时兜底: 整个 body 剥标签
    if not out:
        txt = re.sub(r'<[^>]+>', '', body)
        txt = re.sub(r'\s+', ' ', txt.replace('\u3000', ' ')).strip()
        if txt:
            out.append(f'<p>{txt}</p>')
    return '\n'.join(out)


def fetch_list(page, page_num):
    url = LIST_URL if page_num == 1 else f'{BASE}{LIST_PATH}index_{page_num}.html'
    page.goto(url, timeout=60000, wait_until='load')
    try:
        page.wait_for_selector('a', timeout=15000)
    except Exception:
        pass
    page.wait_for_timeout(2500)
    html = page.content()
    items = re.findall(r'<a[^>]+href="(/zhuzhan/tszwgktzgg/(\d{8})/(\d+)\.html)"[^>]*>([^<]{4,80})</a>', html)
    results = []
    for href, datestr, iid, title in items:
        title = re.sub(r'\s+', ' ', title).strip()
        if not title or len(title) < 4:
            continue
        results.append({
            'url': BASE + href,
            'pub_date': f'{datestr[:4]}-{datestr[4:6]}-{datestr[6:8]}',
            'title': title,
        })
    # 去重 (同页可能重复)
    seen = set()
    uniq = []
    for r in results:
        if r['url'] not in seen:
            seen.add(r['url'])
            uniq.append(r)
    return uniq


def fetch_detail(page, rec):
    try:
        page.goto(rec['url'], timeout=60000, wait_until='load')
        try:
            page.wait_for_selector('div#conN', timeout=15000)
        except Exception:
            pass
        page.wait_for_timeout(2000)
        html = page.content()
    except Exception as e:
        print(f'  [WARN] 详情抓取失败: {rec["url"]} - {e}', file=sys.stderr)
        return None
    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = m.group(1).strip() if m else rec['title']
    m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    pub_date = m.group(1)[:10] if m else rec['pub_date']
    content = clean_body(html, rec['url'])
    if not content.strip():
        return None
    return {
        'title': title,
        'pub_date': pub_date,
        'content': content,
        'site_name': SITE_NAME,
        'source_url': rec['url'],
        'url': rec['url'],
        'group_name': GROUP_NAME,
    }


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=7)
    ap.add_argument('--max-detail', type=int, default=0, help='调试: 只抓前N条详情')
    args = ap.parse_args()

    from playwright.sync_api import sync_playwright
    total_pages = args.pages
    print(f'=== {SITE_NAME} === 列表页数: {total_pages}, CUTOFF: {CUTOFF}')

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=['--disable-blink-features=AutomationControlled'])
        ctx = browser.new_context(user_agent=UA, locale='zh-CN')
        page = ctx.new_page()

        # Step 1: 列表
        all_recs = []
        for pg in range(1, total_pages + 1):
            try:
                recs = fetch_list(page, pg)
                print(f'  P{pg}: {len(recs)} 条')
            except Exception as e:
                print(f'  [WARN] P{pg} 失败: {e}')
                recs = []
            if not recs:
                # 越界页 (P8) 停止
                if pg > 1:
                    print(f'  P{pg} 空 → 实际末页 P{pg-1}')
                    break
            all_recs.extend(recs)
        print(f'列表合计: {len(all_recs)} 条')

        # CUTOFF 过滤 (窗口外跳过)
        in_window = [r for r in all_recs if r['pub_date'] >= CUTOFF]
        out_window = len(all_recs) - len(in_window)
        print(f'窗口内: {len(in_window)}, 窗口外(2023-08-09前): {out_window}')
        if not in_window:
            print('[RESULT] 无窗口内数据')
            browser.close()
            return

        # Step 2: 详情
        records = []
        for i, rec in enumerate(in_window):
            if args.max_detail and i >= args.max_detail:
                break
            d = fetch_detail(page, rec)
            if d:
                records.append(d)
            if (i + 1) % 10 == 0:
                print(f'  [Detail] {i+1}/{len(in_window)} done')
            time.sleep(0.5)
        print(f'\nDetail: {len(records)} success')
        browser.close()

    # Step 3: 入库
    if records:
        push_to_searchdb(records, 'tszwgktzgg')
    print(f'[RESULT] {SITE_NAME}: {len(records)} items')


if __name__ == '__main__':
    main()
