#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""qc_https_scan.py —— 扫「脚本用 http:// 而站点已需 https」的清单（只读）"""
import os, re, json, ssl, time
from concurrent.futures import ThreadPoolExecutor
D = "/root/gov_crawler/"
ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/141.0.0.0 Safari/537.36"}
pairs = {}
for f in sorted(os.listdir(D)):
    if not (f.startswith("crawl_") and f.endswith(".py")): continue
    try: src = open(D + f, encoding="utf-8", errors="ignore").read()
    except Exception: continue
    for u in set(re.findall(r"[\"'](http://[^\s\"'<>()]+)", src)):
        h = u.split("/")[2]
        if h in ("127.0.0.1", "localhost") or h.replace(".", "").isdigit(): continue
        pairs.setdefault(u, set()).add(f)
print("待测 URL 数: %d" % len(pairs))
def probe(u):
    import urllib.request, urllib.error
    out = {}
    for sch in ("http", "https"):
        v = sch + u[4:]
        try:
            r = urllib.request.urlopen(urllib.request.Request(v, headers=UA), timeout=8, context=ctx)
            b = r.read(400000)
            out[sch] = (r.status, len(b))
        except urllib.error.HTTPError as e: out[sch] = (e.code, 0)
        except Exception: out[sch] = (0, 0)
    return u, out
res = []
t0 = time.time()
with ThreadPoolExecutor(max_workers=10) as ex:
    for i, (u, o) in enumerate(ex.map(probe, list(pairs.keys())), 1):
        H, S = o.get("http", (0, 0)), o.get("https", (0, 0))
        cls = "http不可用_https可用" if (S[1] > 2000 and H[1] < 2000) else \
              "都可用" if (H[1] > 2000 and S[1] > 2000) else \
              "都不可用" if (H[1] < 2000 and S[1] < 2000) else "http可用_https差"
        res.append(dict(url=u, http=H, https=S, 判定=cls, 脚本=sorted(pairs[u])[:3]))
        if i % 40 == 0: print("  ...%d/%d (%.0fs)" % (i, len(pairs), time.time() - t0), flush=True)
from collections import Counter
print("\n=== 汇总 ===", Counter(r["判定"] for r in res).most_common())
need = [r for r in res if r["判定"] == "http不可用_https可用"]
print("\n=== 该类 URL（脚本用 http，站点需 https）%d 条 ===" % len(need))
for r in need[:40]:
    print("  %-52s http=%s https=%s | %s" % (r["url"][:52], r["http"], r["https"], ",".join(r["脚本"])[:40]))
json.dump(res, open(D + "qc_out/qc_https_scan.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("\n已写 qc_out/qc_https_scan.json")
