#!/usr/bin/env python3
import json, os, re
D = "/root/gov_crawler/"
d = json.load(open(D + "qc_out/qc_dead_sites2.json", encoding="utf-8"))
hosts = set(r["host"] for r in d)
scs = set()
for f in sorted(os.listdir(D)):
    if not (f.startswith("crawl_") and f.endswith(".py")): continue
    try: src = open(f if os.path.isabs(f) else D + f, encoding="utf-8", errors="ignore").read()
    except Exception: continue
    for u in set(re.findall(r"[\"'](https?://[^\s\"'<>(){}]+)", src)):
        try: h = u.split("/")[2].lower()
        except Exception: continue
        if h in hosts: scs.add(f)
json.dump(sorted(scs), open(D + "qc_out/dead_scripts.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("死站脚本 %d 个" % len(scs))
print(", ".join(sorted(scs)))
