#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""dis_url_lookup.py —— 按脚本名从 Dis 反查（编号 + Url），与死站清单 join"""
import json, os, re, sys, urllib.request
D = "/root/gov_crawler/"; DIS = "22e399b8-3c5b-80f5-b6dc-c6d96cafa1bd"
# 读 token（从 env 或 .env，不硬编码）
tok = os.environ.get("NOTION_API_KEY_HERMES") or os.environ.get("NOTION_TOKEN") or ""
if not tok:
    for p in ("/root/.hermes/.env", os.path.expanduser("~/.hermes/.env"), "/root/gov_crawler/.env"):
        if os.path.exists(p):
            for ln in open(p, encoding="utf-8", errors="ignore"):
                m = re.match(r"\s*(NOTION_API_KEY_HERMES|NOTION_TOKEN|NOTION_API_KEY)\s*=\s*(\S+)", ln)
                if m: tok = m.group(2).strip().strip('"').strip("'"); break
        if tok: break
print("token:", "✅ 已读取" if tok else "❌ 未找到")
if not tok: sys.exit(1)
def q(body):
    req = urllib.request.Request("https://api.notion.com/v1/databases/%s/query" % DIS,
                                 data=json.dumps(body).encode(),
                                 headers={"Authorization": "Bearer " + tok, "Notion-Version": "2022-06-28",
                                          "Content-Type": "application/json"}, method="POST")
    return json.load(urllib.request.urlopen(req, timeout=40))
rows, cur = [], None
while True:
    b = {"page_size": 100}
    if cur: b["start_cursor"] = cur
    r = q(b)
    for p in r.get("results", []):
        pr = p.get("properties", {})
        def txt(k):
            v = pr.get(k, {})
            t = v.get("title") or v.get("rich_text") or []
            return "".join(x.get("plain_text", "") for x in t)
        rows.append(dict(script=txt("Script_name"), no=txt("编号"), url=txt("Url") or txt("URL"), name=txt("名称")))
    if not r.get("has_more"): break
    cur = r.get("next_cursor")
print("Dis 拉取: %d 条" % len(rows))
m = {}
for r in rows:
    if r["script"]: m.setdefault(r["script"].strip(), []).append(r)
d = json.load(open(D + "qc_out/qc_dead_sites2.json", encoding="utf-8"))
targets = json.load(open(D + "qc_out/dead_scripts.json", encoding="utf-8")) if os.path.exists(D + "qc_out/dead_scripts.json") else None
scs = sorted(set(x for x in (targets or []) ))
if not scs:
    print("（缺 dead_scripts.json，改用域名里含脚本名的方式跳过）")
out = []
for s in scs:
    hit = m.get(s) or m.get(s.replace(".py", ""))
    out.append(dict(script=s, dis=hit[:1]))
json.dump(out, open(D + "qc_out/dead_dis_join.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
for x in out[:40]:
    h = x["dis"][0] if x["dis"] else None
    print("  %-30s %s %s" % (x["script"][:30], (h.get("no") if h else "—") or "—",
                              (" | " + (h.get("url") or "")[:56]) if h else "（Dis 未命中）"))
