#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""create_qc_page.py —— 建 QC 结果 Notion 页（父页由 Dis-db 所在块反查）"""
import json
import subprocess

TOK = "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"
DIS_BLOCK = "3d6399b8-3c5b-80c0-8a6f-e65ba39653b9"   # Dis-db 所在块


def api(url, method="GET", body=None):
    cmd = ["curl", "-sS", "-X", method, "-H", "Authorization: Bearer " + TOK,
           "-H", "Notion-Version: 2022-06-28", "-H", "Content-Type: application/json",
           "--max-time", "60", url]
    if body is not None:
        cmd += ["-d", json.dumps(body, ensure_ascii=False)]
    p = subprocess.run(cmd, capture_output=True, timeout=90)
    try:
        return json.loads(p.stdout.decode("utf-8", "ignore"))
    except Exception:
        return {"_raw": p.stdout.decode("utf-8", "ignore")[:300]}


# 父页：块反查不可用（集成读不到该块）→ 挂到已知可访问的 Hermes Memory 页下，用户可后续拖到别处
import sys
PARENT = next((a.split("=",1)[1] for a in sys.argv if a.startswith("--parent=")),
                "35d399b8-3c5b-80b4-90bc-e7d6d10b2f8c")
pg = api("https://api.notion.com/v1/pages/%s" % PARENT)
t = "".join(x.get("plain_text", "") for x in (pg.get("properties", {}).get("title", {}).get("title") or []))
print("父页:", t or "(读不到标题)", PARENT)
parent = PARENT


def h2(s, lvl="heading_2"):
    return {"object": "block", "type": lvl, "lvl" if False else lvl: {"rich_text": [{"type": "text", "text": {"content": s}}]}}


def bul(s):
    return {"object": "block", "type": "bulleted_list_item",
            "bulleted_list_item": {"rich_text": [{"type": "text", "text": {"content": s}}]}}


def par(s):
    return {"object": "block", "type": "paragraph",
            "paragraph": {"rich_text": [{"type": "text", "text": {"content": s}}]}}


blocks = [
    h2("一、口径（2026-09-25 确认）"),
    bul("主键 = Script_name。配置编号已弃用：历史遗留（当时调度=50 个/批分批跑，编号=当时数组快照序），现行调度不看它 → QC 不校验编号。"),
    bul("停滞 > 30 天 算异常。"),
    bul("先列清单，再定批量策略；Tier1 不改任何数据。"),
    bul("三层：Tier1 全自动清单（覆盖 100%，只列异常）→ Tier2 人工 10 个/批只处理异常 → Tier3 抽样反验判据。"),
    h2("二、Tier 1 首跑结果（2026-09-25）"),
    par("config 2114 条 | gov_raw 580,796 行 | script_name 空 3,669（0.63%）"),
    bul("D2 孤儿脚本（磁盘有、config 无 → 永不更新）：43"),
    bul("D3 不支持 --pages：522（判据待收紧）"),
    bul("D4 库内脚本名磁盘查无此文件：62 个 / 13,876 行（含 luan、jz、kmyl 等非文件名垃圾值）"),
    bul("D5 config 注册但库内 0 行：190（空跑）"),
    bul("D6 库有数据但不在 config：22（改名/退役）"),
    bul("D7 停滞 >30 天：495（30-90d 254 / 90-180d 124 / 180-365d 45 / >365d 64）"),
    bul("D8 行数≥30 却全无附件：1033（linkbox 类 bug 高危候选池，非全是 bug）"),
    bul("D9 存在空正文行：302 个脚本"),
    bul("D10 FTS 计数：gov_search 634,551 vs gov_raw 580,796（+53,755，需先确认 gov_search 服务范围）"),
    {"object": "block", "type": "heading_3", "heading_3": {"rich_text": [
        {"type": "text", "text": {"content": "D1 假阳性说明（自我纠正，重要）"}}]}},
    par("Tier1 v1 把「args 为空」(718) 与「sync_mode 非 direct_server」(432) 判为异常 —— 错：调度器对二者都有默认值（日增量默认 1 页是正常设计）。故「有问题的脚本 1917/2034」是虚数。加判据前必须先用 crawl_scheduler_v2.py 的真实解析行为校准。"),
    h2("三、待裁定清单（先列，不批量动）"),
    bul("D1 判据收紧（剔除 args/sync_mode 假阳性）—— 1150 条"),
    bul("D3 判据收紧 —— 522 个需逐个确认"),
    bul("(script,args) 完全重复 → 会重复跑 —— 1 个（crawl_changshu_gsgg.py）"),
    bul("孤儿脚本 43 个：注册 or 归档"),
    bul("库内脚本名查无文件 —— 62 个 / 13,876 行"),
    bul("注册但 0 行 —— 190 个：修 or 注销"),
    bul("停滞 —— 495 个：逐站先证源站再判爬虫"),
    bul("疑似无附件 —— 1033 个：页面级抽样核查"),
    bul("空正文 —— 302 个脚本"),
    bul("空 script_name —— 3,669 行是否可填充"),
    bul("FTS 计数差 +53,755：确认 gov_search 服务范围"),
    h2("四、批次记录（Tier 2，10 个/批）"),
    par("（等 Tier1 判据收紧后开始）"),
    h2("五、明细文件"),
    par("全量清单在本地 ~/Crawler/gov_crawler/qc_out/qc_tier1_20260925.md / .json 及 QC结果文档.md（三处同步）。"),
]

r = api("https://api.notion.com/v1/pages", "POST", {
    "parent": {"page_id": parent},
    "properties": {"title": {"title": [{"type": "text", "text": {"content": "爬虫数据 QC 结果（活文档）"}}]}},
    "children": blocks})
if r.get("id"):
    print("✅ 已建页:", r.get("url"))
    back = api("https://api.notion.com/v1/pages/%s" % r["id"])
    bt = "".join(x.get("plain_text", "") for x in (back.get("properties", {}).get("title", {}).get("title") or []))
    ch = api("https://api.notion.com/v1/blocks/%s/children?page_size=100" % r["id"])
    print("   回读标题:", bt, "| 块数:", len(ch.get("results", [])))
else:
    print("❌ 失败:", json.dumps(r, ensure_ascii=False)[:400])
