#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""wenshang 前5页详情加速抓取 — 列表用 playwright(原脚本), 详情用 urllib 直抓
用法: python3 wenshang_detail_fast.py
读列表(playwright 85条) → 详情(urllib, IP+Host) → /tmp/wenshang_gggs.jsonl (断点续传)
"""
import json, os, re, sys, time, random, urllib.request
from datetime import datetime
from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

BASE = "http://www.wenshang.gov.cn"
LIST_URL = BASE + "/col/col61746/index.html?vc_xxgkarea=1137083000433565XHA&number=A0002001&jh=263"
IP = "27.221.53.71"
OUT = "/tmp/wenshang_gggs.jsonl"
MAX_PAGES = 5
SLEEP_MIN, SLEEP_MAX = 0.3, 0.6
MAX_FAIL = 8
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"

sys.path.insert(0, "/root/gov_crawler")
import importlib.util
spec = importlib.util.spec_from_file_location("cwg", "/root/gov_crawler/crawl_wenshang_gggs.py")
cwg = importlib.util.module_from_spec(spec)
spec.loader.exec_module(cwg)
parse_list = cwg.parse_list
parse_detail = cwg.parse_detail
KEEP = cwg.KEEP
UNWRAP = cwg.UNWRAP
DROP = cwg.DROP
clean_html = cwg.clean_html

def fetch_detail(url):
    """urllib 直抓详情页: IP + Host 头"""
    real = url.replace(BASE, f"http://{IP}")
    req = urllib.request.Request(real, headers={
        "User-Agent": UA,
        "Host": "www.wenshang.gov.cn",
        "Accept": "text/html,application/xhtml+xml,*/*;q=0.8",
        "Accept-Language": "zh-CN,zh;q=0.9",
        "Connection": "close",
    })
    with urllib.request.urlopen(req, timeout=20) as r:
        return r.read().decode("utf-8", errors="ignore")

def main():
    max_pages = MAX_PAGES
    done_ids = set()
    if os.path.exists(OUT):
        for line in open(OUT, encoding="utf-8"):
            try:
                done_ids.add(json.loads(line)["id"])
            except Exception:
                pass
    print(f"[*] 已有 {len(done_ids)} 条, 断点续传", flush=True)

    # 1. playwright 抓列表前5页
    all_items = []
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled", "--no-sandbox",
            f"--host-resolver-rules=MAP www.wenshang.gov.cn {IP}, MAP wenshang.gov.cn {IP}",
        ])
        ctx = browser.new_context(user_agent=UA, locale="zh-CN", viewport={"width": 1920, "height": 1080})
        page = ctx.new_page()
        try:
            page.goto(LIST_URL, timeout=30000, wait_until="domcontentloaded")
        except Exception:
            pass
        passed = False
        for i in range(12):
            time.sleep(2)
            try:
                html = page.content()
                if "zfxxgk_item" in html and "共" in html:
                    passed = True
                    break
            except Exception:
                pass
        if not passed:
            print("[X] 列表未渲染", flush=True)
            sys.exit(1)
        print("[✓] 列表已渲染", flush=True)
        fail = 0
        for pno in range(1, max_pages + 1):
            html = page.content()
            items = parse_list(html)
            if not items:
                fail += 1
                if fail >= 3:
                    break
            else:
                fail = 0
            new_items = [it for it in items if it["id"] not in done_ids]
            all_items.extend(items)
            print(f"[list] p{pno}: {len(items)} 条(新{len(new_items)}), 累计 {len(all_items)}", flush=True)
            # 下一页
            try:
                page.evaluate(f"funGoPage('/module/xxgk/search.jsp', {pno + 1})")
            except Exception:
                pass
            time.sleep(random.uniform(1.0, 1.8))
        browser.close()
    print(f"[*] 列表共 {len(all_items)} 条, 开始抓详情", flush=True)

    # 2. urllib 直抓详情 (断点续传)
    f = open(OUT, "a", encoding="utf-8")
    ok = fail = 0
    new_items = [it for it in all_items if it["id"] not in done_ids]
    for it in new_items:
        try:
            html = fetch_detail(it["url"])
        except Exception as e:
            print(f"[!] 详情 {it['id']} 抓取失败: {type(e).__name__}", flush=True)
            fail += 1
            if fail >= MAX_FAIL:
                print("[!] 失败过多, 中止", flush=True)
                break
            continue
        det = parse_detail(html)
        if not det["body"]:
            fail += 1
            if fail >= MAX_FAIL:
                print("[!] 空正文过多, 中止", flush=True)
                break
            print(f"[warn] 空正文 {it['id']} {it['title'][:30]}", flush=True)
        else:
            fail = 0
        rec = {
            "id": it["id"],
            "title": det["title"] or it["title"],
            "date": det["date"] or it["date"],
            "url": it["url"],
            "source": "wenshang",
            "site": "wenshang_gggs",
            "body": det["body"],
            "attachments": det["attachments"],
            "crawl_time": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
        }
        f.write(json.dumps(rec, ensure_ascii=False) + "\n")
        f.flush()
        done_ids.add(it["id"])
        ok += 1
        if ok % 20 == 0:
            print(f"[detail] +{ok} 条 (累计 {len(done_ids)})", flush=True)
        time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
    f.close()
    print(f"[✓] 完成: 本次新增 {ok} 条, 总计 {len(done_ids)} 条 -> {OUT}", flush=True)

if __name__ == "__main__":
    main()
