#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""qc_dead_sites2.py —— 死站判定 v2：URL 归一化 + 要求两个不同深路径 + 占位标记判据"""
import hashlib, json, os, re, ssl, time, urllib.request, urllib.error
from collections import defaultdict, Counter
from concurrent.futures import ThreadPoolExecutor
D = "/root/gov_crawler/"
ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/141.0.0.0 Safari/537.36"}
def norm(u):
    u = re.sub(r"^http://", "https://", u.strip())
    m = re.match(r"(https://[^/]+)(/.*)?$", u)
    if not m: return u
    host, path = m.group(1).lower(), (m.group(2) or "/")
    if len(path) > 1 and path.endswith("/"): path = path[:-1]
    return host + path
hosts = defaultdict(set)
for f in sorted(os.listdir(D)):
    if not (f.startswith("crawl_") and f.endswith(".py")): continue
    try: src = open(D + f, encoding="utf-8", errors="ignore").read()
    except Exception: continue
    for u in set(re.findall(r"[\"'](https?://[^\s\"'<>(){}]+)", src)):
        try: h = u.split("/")[2].lower()
        except Exception: continue
        if h.replace(".", "").isdigit() or "localhost" in h or "example.com" in h: continue
        hosts[h].add(norm(u))
print("域名数: %d" % len(hosts), flush=True)
def probe(u):
    try:
        r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=9, context=ctx)
        b = r.read(500000)
        return r.status, len(b), hashlib.md5(b).hexdigest(), len(re.findall(rb"<a\s", b)), len(re.findall(rb"<li", b))
    except urllib.error.HTTPError as e: return e.code, 0, "", 0, 0
    except Exception: return 0, 0, "", 0, 0
def check(h):
    root = "https://" + h
    cand = sorted(x for x in hosts[h] if x.rstrip("/") != root)
    urls = ([root] + cand[:2]) if cand else [root]
    if len(set(urls)) < 2: return h, None
    out = [probe(u) for u in urls[:3]]
    return h, list(zip(urls[:3], out))
res = []
t0 = time.time()
with ThreadPoolExecutor(max_workers=8) as ex:
    for i, (h, out) in enumerate(ex.map(check, list(hosts.keys())), 1):
        if out:
            good = [(u, o) for u, o in out if o[0] == 200 and o[1] > 0]
            same = len(good) >= 2 and len(set(o[2] for _, o in good)) == 1
            place = bool(good) and all(o[3] < 5 and o[4] < 5 and o[1] < 6000 for _, o in good)
            if same or place:
                res.append(dict(host=h, 判定="不同路径同内容" if same else "占位页(无列表标记)",
                                详情=[(u[:58], o[0], o[1]) for u, o in out]))
        if i % 100 == 0: print("  ...%d/%d (%.0fs)" % (i, len(hosts), time.time() - t0), flush=True)
json.dump(res, open(D + "qc_out/qc_dead_sites2.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("\n=== v2 结果: %d 个 ===" % len(res))
print("  判定:", Counter(r["判定"] for r in res).most_common())
for r in res[:25]:
    print("   %-32s %-16s %s" % (r["host"][:32], r["判定"], r["详情"][:2]))
print("ALLDONE")
