#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""qc_dead_sites.py —— 找「站点已死」：同域名多个 URL 返回同一份内容（只读）"""
import hashlib, json, os, re, ssl, time, urllib.request, urllib.error
from collections import defaultdict
from concurrent.futures import ThreadPoolExecutor
D = "/root/gov_crawler/"
ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/141.0.0.0 Safari/537.36"}
# 收集 域名 -> 代表 URL（每域名最多 2 条）
hosts = defaultdict(set)
for f in sorted(os.listdir(D)):
    if not (f.startswith("crawl_") and f.endswith(".py")): continue
    try: src = open(D + f, encoding="utf-8", errors="ignore").read()
    except Exception: continue
    for u in set(re.findall(r"[\"'](https?://[^\s\"'<>(){}]+)", src)):
        try: h = u.split("/")[2].lower()
        except Exception: continue
        if h.replace(".", "").isdigit() or "localhost" in h: continue
        hosts[h].add(u)
print("域名数: %d" % len(hosts), flush=True)
def probe(u):
    try:
        r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=8, context=ctx)
        b = r.read(400000)
        return r.status, len(b), hashlib.md5(b).hexdigest()
    except urllib.error.HTTPError as e: return e.code, 0, ""
    except Exception: return 0, 0, ""
def check(h):
    urls = ["https://" + h + "/"] + sorted(hosts[h])[:2]
    out = []
    for u in urls[:3]:
        st, ln, md5 = probe(u)
        out.append((u, st, ln, md5))
        time.sleep(0.2)
    return h, out
res = []
t0 = time.time()
with ThreadPoolExecutor(max_workers=8) as ex:
    for i, (h, out) in enumerate(ex.map(check, list(hosts.keys())), 1):
        good = [x for x in out if x[1] == 200 and x[2] > 0]
        same = len(set(x[3] for x in good)) == 1 and len(good) >= 2
        tiny = all(x[2] < 4000 for x in good) if good else False
        if same or (tiny and good):
            res.append(dict(host=h, 判定="同内容" if same else "内容过小", 详情=[(x[0][:60], x[1], x[2]) for x in out]))
        if i % 50 == 0: print("  ...%d/%d (%.0fs)" % (i, len(hosts), time.time() - t0), flush=True)
json.dump(res, open(D + "qc_out/qc_dead_sites.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("\n=== 结果：疑似已死/占位站点 %d 个 ===" % len(res))
from collections import Counter
print("  判定分布:", Counter(r["判定"] for r in res).most_common())
for r in res[:30]:
    print("   %-34s %-8s %s" % (r["host"][:34], r["判定"], r["详情"][:2]))
print("ALLDONE")
