#!/usr/bin/env python3
"""crawl_wenshang_gggs.py — 汶上县人民政府 公告公示 (wenshang.gov.cn/col/col61746)
大汉版通 + 山东xxgk模块, search.jsp 列表需要 JS 渲染 → playwright headless
列表: div.zfxxgk_item > ul > li > a[title] + b{日期}; 共2669条/157页
分页: 点击"下一页" funGoPage('/module/xxgk/search.jsp', N) JS局部刷新
详情: /art/{年}/{月}/{日}/art_61746_{ID}.html?xxgkhide=1; 正文 div#zoom.main-txt
"""
import json, os, re, sys, time, random
from datetime import datetime
from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

BASE = "http://www.wenshang.gov.cn"
LIST_URL = BASE + "/col/col61746/index.html?vc_xxgkarea=1137083000433565XHA&number=A0002001&jh=263"
IP = "27.221.53.71"
OUT = "/tmp/wenshang_gggs.jsonl"
MAX_PAGES = 160
SLEEP_MIN, SLEEP_MAX = 0.8, 1.6
MAX_FAIL = 8

KEEP = {'p', 'table', 'tr', 'td', 'th', 'thead', 'tbody', 'a', 'img', 'br', 'h1', 'h2', 'h3', 'h4', 'ul', 'ol', 'li', 'strong', 'em', 'b'}
UNWRAP = {'span', 'div', 'font', 'label', 'section', 'article'}
DROP = {'script', 'style', 'iframe', 'form', 'input', 'button', 'nav', 'header', 'footer'}

def clean_html(html):
    if not html:
        return ""
    # 清 jcms/ZJEG 注释 + meta
    html = re.sub(r'<!--.*?-->', '', html, flags=re.S)
    html = re.sub(r'<meta[^>]*>', '', html, flags=re.I)
    soup = BeautifulSoup(html, 'html.parser')
    for tag in soup.find_all(True):
        if tag.name in DROP:
            tag.decompose()
            continue
        if tag.name == 'img':
            src = tag.get('src') or tag.get('data-src') or ''
            if src:
                if src.startswith('/'):
                    src = BASE + src
                a = soup.new_tag('a')
                a['href'] = src
                a.string = '查看图片'
                tag.replace_with(a)
            else:
                tag.decompose()
            continue
        if tag.name == 'a':
            href = tag.get('href') or ''
            if href.startswith('/'):
                tag['href'] = BASE + href
            elif href and not href.startswith('http'):
                tag['href'] = BASE + '/' + href.lstrip('./')
        for attr in list(tag.attrs):
            if attr.startswith(('on', 'style', 'class', 'id', 'aria-', 'data-', 'target', 'rel', 'align', 'width', 'height', 'border', 'cellpadding', 'cellspacing', 'valign', 'scope', 'rowspan', 'colspan')):
                del tag[attr]
    for tag in soup.find_all(UNWRAP):
        tag.unwrap()
    for tag in soup.find_all(['td', 'th']):
        if not tag.find_parent('table'):
            tag.unwrap()
    out = str(soup)
    out = re.sub(r'\n{3,}', '\n\n', out)
    out = re.sub(r'[ \t]{2,}', ' ', out)
    return out.strip()

def parse_list(html):
    """列表: div.zfxxgk_item li > a[title] + b{date}"""
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for li in soup.select('div.zfxxgk_item li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        m = re.search(r'/art/\d{4}/\d{1,2}/\d{1,2}/art_\d+_(\d+)\.html', href)
        if not m:
            continue
        title = a.get('title') or a.get_text(strip=True) or ''
        title = re.sub(r'^【[^】]*】', '', title).strip()  # 去【公告公示】前缀
        b = li.find('b')
        date = b.get_text(strip=True) if b else ''
        dm = re.search(r'(\d{4})-(\d{2})-(\d{2})', date)
        date = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}" if dm else ''
        items.append({
            'id': m.group(1),
            'title': title.strip(),
            'date': date,
            'url': href if href.startswith('http') else BASE + href,
        })
    seen, out = set(), []
    for it in items:
        if it['url'] not in seen:
            seen.add(it['url'])
            out.append(it)
    return out

def parse_detail(html):
    soup = BeautifulSoup(html, 'html.parser')
    def meta(name):
        m = soup.find('meta', attrs={'name': name})
        return (m.get('content') or '').strip() if m else ''
    title = meta('ArticleTitle') or ''
    pub = meta('pubdate') or meta('PubDate') or ''
    zoom = soup.find('div', id='zoom') or soup.find('div', class_='main-txt') \
        or soup.find('div', class_='TRS_Editor') or soup.find('div', class_='content')
    body = ''
    if zoom:
        body = clean_html(str(zoom))
    dm = re.search(r'(\d{4})-(\d{1,2})-(\d{1,2})', pub)
    date = f"{dm.group(1)}-{int(dm.group(2)):02d}-{int(dm.group(3)):02d}" if dm else ''
    # 附件
    attachments = []
    if zoom:
        for a in zoom.find_all('a', href=True):
            h = a['href'].lower()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj|wps|et)$', h):
                attachments.append({'name': a.get_text(strip=True) or '附件', 'url': a['href']})
    return {'title': title, 'body': body, 'date': date, 'attachments': attachments}

def main():
    max_pages = int(sys.argv[1]) if len(sys.argv) > 1 else MAX_PAGES
    done_ids = set()
    if os.path.exists(OUT):
        for line in open(OUT, encoding='utf-8'):
            try:
                done_ids.add(json.loads(line)['id'])
            except Exception:
                pass
    print(f"[*] 已有 {len(done_ids)} 条, 断点续传", flush=True)

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled", "--no-sandbox",
            f"--host-resolver-rules=MAP www.wenshang.gov.cn {IP}, MAP wenshang.gov.cn {IP}",
        ])
        ctx = browser.new_context(
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
            locale="zh-CN", viewport={"width": 1920, "height": 1080},
        )
        page = ctx.new_page()
        try:
            page.goto(LIST_URL, timeout=30000, wait_until="domcontentloaded")
        except Exception as e:
            print(f"[!] goto exc: {type(e).__name__}", flush=True)
        # 等列表渲染
        passed = False
        for i in range(12):
            time.sleep(2)
            try:
                html = page.content()
                if 'zfxxgk_item' in html and '共' in html:
                    passed = True
                    break
            except Exception:
                pass
        if not passed:
            print("[X] 列表未渲染", flush=True)
            browser.close()
            sys.exit(1)
        print("[✓] 列表已渲染", flush=True)

        # ---- 抓列表 157 页 (点击下一页) ----
        all_items = []
        fail = 0
        for pno in range(1, max_pages + 1):
            html = page.content()
            items = parse_list(html)
            if not items:
                fail += 1
                if fail >= MAX_FAIL:
                    print(f"[!] 连续 {MAX_FAIL} 页无条目, 中止", flush=True)
                    break
            else:
                fail = 0
                seen_ids = {it['id'] for it in all_items}
                new_items = [it for it in items if it['id'] not in seen_ids]
                all_items.extend(new_items)
                print(f"[list] p{pno}: {len(items)} 条(新{len(new_items)}), 累计 {len(all_items)}", flush=True)
            # 是否有下一页
            if pno >= max_pages:
                break
            try:
                page.click("text=下一页")
            except Exception as e:
                print(f"[!] 下一页点击失败 p{pno}: {e}", flush=True)
                fail += 1
                if fail >= MAX_FAIL:
                    break
                continue
            time.sleep(random.uniform(1.2, 2.0))

        print(f"[*] 列表共 {len(all_items)} 条, 开始抓详情", flush=True)

        # ---- 抓详情 ----
        f = open(OUT, 'a', encoding='utf-8')
        ok = 0
        fail = 0
        for it in all_items:
            if it['id'] in done_ids:
                continue
            try:
                page.goto(it['url'], timeout=25000, wait_until="domcontentloaded")
            except Exception as e:
                pass  # domcontentloaded 被外部资源阻塞, 超时后继续读 content
            try:
                time.sleep(1.5)
                html = page.content()
            except Exception as e:
                print(f"[!] 详情 {it['id']} content 失败: {e}", flush=True)
                fail += 1
                if fail >= MAX_FAIL:
                    print("[!] 失败过多, 中止 (下次断点续传)", flush=True)
                    break
                continue
            det = parse_detail(html)
            if not det['body']:
                fail += 1
                if fail >= MAX_FAIL:
                    break
                print(f"[warn] 空正文 {it['id']} {it['title'][:30]}", flush=True)
            else:
                fail = 0
            rec = {
                'id': it['id'],
                'title': det['title'] or it['title'],
                'date': det['date'] or it['date'],
                'url': it['url'],
                'source': 'wenshang',
                'site': 'wenshang_gggs',
                'body': det['body'],
                'attachments': det['attachments'],
                'crawl_time': datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
            }
            f.write(json.dumps(rec, ensure_ascii=False) + '\n')
            f.flush()
            done_ids.add(it['id'])
            ok += 1
            if ok % 30 == 0:
                print(f"[detail] +{ok} 条 (累计 {len(done_ids)})", flush=True)
            time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
        f.close()
        browser.close()
        print(f"[✓] 完成: 本次新增 {ok} 条, 总计 {len(done_ids)} 条 -> {OUT}", flush=True)

if __name__ == '__main__':
    main()
