#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""crawl_yanshou100_daily.py - yanshou100 日跑包装: 抓取 + 导入一体

解决日跑"只抓不导入/--out 覆盖/无去重"三个问题:
1. --type 区分栏目, --out 各自独立文件 (yanshou/qita 互不覆盖)
2. 自动从 DB 读已有 itemId 作为 --existing 去重, 只抓新增
3. 抓完立即 import_jsonl_raw.py 导入 (保留 HTML/表格/附件)

用法: python3 crawl_yanshou100_daily.py --type yanshou|qita [--pages N]
"""
import argparse, os, re, sqlite3, subprocess, sys, json

BASE_DIR = "/root/gov_crawler"
DB = os.getenv("SEARCH_DB", "/root/search.db")

def get_existing_ids():
    """从 gov_raw 提取已有 itemId (page_url 的 id= 参数)"""
    ids = set()
    try:
        conn = sqlite3.connect(DB, timeout=60)
        for (u,) in conn.execute(
            "SELECT DISTINCT page_url FROM gov_raw WHERE page_url LIKE '%item_detail.html?id=%'"
        ):
            m = re.search(r"id=(\d+)", u or "")
            if m:
                ids.add(m.group(1))
        conn.close()
    except Exception as e:
        print(f"  [WARN] 读已有ID失败: {e}", flush=True)
    return ids

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--type", default="yanshou", help="yanshou/qita/jiance/fangan")
    ap.add_argument("--pages", type=int, default=5)
    args = ap.parse_args()

    existing = get_existing_ids()
    existing_file = f"/tmp/ys100_existing_{args.type}.txt"
    with open(existing_file, "w", encoding="utf-8") as f:
        for i in sorted(existing):
            f.write(i + "\n")
    print(f"已有 itemId: {len(existing)} 个 → {existing_file}", flush=True)

    out_file = f"/tmp/ys100_{args.type}.jsonl"
    # 清理旧文件 (上次残留)
    if os.path.exists(out_file):
        os.remove(out_file)

    # 1. 抓取
    crawl_cmd = [
        sys.executable, os.path.join(BASE_DIR, "crawl_yanshou100.py"),
        "--pages", str(args.pages),
        "--type", args.type,
        "--out", out_file,
        "--existing", existing_file,
    ]
    print(f"▶ 抓取: {' '.join(crawl_cmd)}", flush=True)
    r = subprocess.run(crawl_cmd, capture_output=True, text=True, timeout=1200)
    print(r.stdout[-2000:] if r.stdout else "", flush=True)
    if r.returncode != 0:
        print(f"✗ 抓取失败 exit={r.returncode} stderr={r.stderr[-500:]}", flush=True)
        sys.exit(1)

    if not os.path.exists(out_file) or os.path.getsize(out_file) == 0:
        print("⚠ JSONL 为空 (无新增或抓取失败), 跳过导入", flush=True)
        return

    n = sum(1 for _ in open(out_file, encoding="utf-8") if _.strip())
    print(f"JSONL: {out_file} ({n} 条)", flush=True)

    # 2. 导入
    imp_cmd = [sys.executable, os.path.join(BASE_DIR, "import_jsonl_raw.py"), out_file]
    print(f"▶ 导入: {' '.join(imp_cmd)}", flush=True)
    r2 = subprocess.run(imp_cmd, capture_output=True, text=True, timeout=600)
    print(r2.stdout[-500:] if r2.stdout else "", flush=True)
    if r2.returncode != 0:
        print(f"✗ 导入失败 exit={r2.returncode} stderr={r2.stderr[-500:]}", flush=True)
        sys.exit(1)

    # 3. 清理
    os.remove(out_file)
    print(f"✅ {args.type} 日跑完成: 抓{n}条 → 已导入", flush=True)

if __name__ == "__main__":
    main()
