#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""qc_d5_recheck2.py —— D5 重测：从脚本源码取目标域名 vs 库内实际域名（只读）"""
import json, os, re, sqlite3
from collections import Counter
D = "/root/gov_crawler/"
db = sqlite3.connect("file:/root/search.db?mode=ro", uri=True, timeout=30)
db.execute("PRAGMA busy_timeout=30000")
hosts = set()
for r in db.execute("SELECT DISTINCT substr(page_url, instr(page_url,'//')+2, "
                    "CASE WHEN instr(substr(page_url, instr(page_url,'//')+2),'/')>0 "
                    "THEN instr(substr(page_url, instr(page_url,'//')+2),'/')-1 ELSE 999 END) FROM gov_raw"):
    h = (r[0] or "").lower().strip()
    if h: hosts.add(h)
print("库内域名数: %d" % len(hosts))
hosts_nw = set(h[4:] if h.startswith("www.") else h for h in hosts)

CFG = json.load(open(D + "daily_crawl_config.json", encoding="utf-8"))
cfg = CFG if isinstance(CFG, list) else CFG.get("tasks", CFG.get("scripts", []))
no_data, no_src, no_host, ok = [], [], [], 0
for c in cfg:
    if not c.get("enabled", True): continue
    s = (c.get("script") or "").strip()
    p = D + s
    if not s.endswith(".py") or not os.path.exists(p):
        no_src.append((s, c.get("name") or "")); continue
    src = open(p, encoding="utf-8", errors="ignore").read()[:60000]
    hs = set()
    for m in re.finditer(r"https?://([A-Za-z0-9._\-]+)", src):
        h = m.group(1).lower()
        if h and "." in h and "w3.org" not in h and "example.com" not in h:
            hs.add(h)
    hs = set(h for h in hs if not h.replace(".", "").isdigit())
    if not hs:
        no_host.append((s, c.get("name") or "")); continue
    hit = any((h in hosts) or (h[4:] in hosts_nw if h.startswith("www.") else h in hosts_nw)
              or any(x.endswith("." + (h[4:] if h.startswith("www.") else h)) for x in hosts_nw) for h in hs)
    if hit: ok += 1
    else: no_data.append((s, c.get("name") or "", sorted(hs)[0][:36]))
print("\n=== D5 重测（启用脚本 %d）===" % sum(1 for c in cfg if c.get("enabled", True)))
print("  域名在库内（有数据）   : %d ✅" % ok)
print("  域名全不在库内（真无数据）: %d ⚠️" % len(no_data))
print("  脚本文件缺失            : %d" % len(no_src))
print("  源码里取不到域名         : %d" % len(no_host))
print("\n=== 真·无数据（最多 18）===")
for s, n, h in no_data[:18]:
    print("  %-32s %-20s %s" % (s[:32], n[:20], h))
json.dump({"ok": ok, "no_data": no_data, "no_src": no_src, "no_host": no_host},
          open(D + "qc_out/qc_d5_recheck_20260925.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("\n已写 qc_out/qc_d5_recheck_20260925.json")
db.close()
