#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""find_notion_parent.py —— 列出集成可见的顶层页面，定位 Crawler 父页"""
import json
import subprocess

TOK = "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"


def api(url, body=None):
    cmd = ["curl", "-sS", "-X", "POST", "-H", "Authorization: Bearer " + TOK,
           "-H", "Notion-Version: 2022-06-28", "-H", "Content-Type: application/json",
           "--max-time", "60", url]
    cmd += ["-d", json.dumps(body if body is not None else {}, ensure_ascii=False)]
    p = subprocess.run(cmd, capture_output=True, timeout=90)
    try:
        return json.loads(p.stdout.decode("utf-8", "ignore"))
    except Exception:
        return {"_raw": p.stdout.decode("utf-8", "ignore")[:300]}


r = api("https://api.notion.com/v1/search", {"page_size": 100})
print("object:", r.get("object"), "| 结果数:", len(r.get("results", [])), "| has_more:", r.get("has_more"))
if "_raw" in r:
    print("原始:", r["_raw"])
top, child = [], []
for pg in r.get("results", []):
    if pg.get("object") != "page":
        continue
    t = "".join(x.get("plain_text", "") for x in (pg.get("properties", {}).get("title", {}).get("title") or []))
    par = pg.get("parent", {})
    (top if par.get("type") == "workspace" else child).append((t, pg["id"], par.get("type"), par.get("page_id", "")))
print("\n=== 顶层页面（parent=workspace）%d 个 ===" % len(top))
for t, i, _, _ in top:
    print("  %-52s %s" % (t[:52], i))
print("\n=== 子页面前 25 个（含关键词的优先）===")
kw = ("爬虫", "Crawler", "日报", "数据")
for t, i, pt, pid in sorted(child, key=lambda x: (not any(k in x[0] for k in kw), x[0]))[:25]:
    mark = "⭐" if any(k in t for k in kw) else "  "
    print("  %s %-46s %s parent_page=%s" % (mark, t[:46], i, pid[:8]))
