#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""dead_url_map.py —— 27 个死站：域名→URL→脚本名（供 Dis 反查）"""
import json, os, re
D = "/root/gov_crawler/"
d = json.load(open(D + "qc_out/qc_dead_sites2.json", encoding="utf-8"))
host2scripts = {}
for f in sorted(os.listdir(D)):
    if not (f.startswith("crawl_") and f.endswith(".py")): continue
    try: src = open(D + f, encoding="utf-8", errors="ignore").read()
    except Exception: continue
    for u in set(re.findall(r"[\"'](https?://[^\s\"'<>(){}]+)", src)):
        try: h = u.split("/")[2].lower()
        except Exception: continue
        host2scripts.setdefault(h, {}).setdefault(f, set()).add(u)
print("%-30s %-16s %s" % ("域名", "判定", "脚本 / URL"))
print("-" * 130)
for r in d:
    h = r["host"]
    scs = host2scripts.get(h, {})
    tag = r["判定"]
    if not scs:
        print("%-30s %-16s (未找到对应脚本)" % (h[:30], tag))
        continue
    for f, urls in list(scs.items())[:2]:
        u0 = sorted(urls)[0]
        print("%-30s %-16s %-30s %s" % (h[:30], tag, f[:30], u0[:70]))
