#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""chk_residual_http.py —— 查残留的 http:// 目标域名引用"""
import os, re
D = "/root/gov_crawler/"
DOMS = ["www.anlu.gov.cn", "www.jxdy.gov.cn", "www.jxln.gov.cn", "www.shaowu.gov.cn",
        "www.xunwu.gov.cn", "www.yantai.gov.cn", "www.zhcqhj.com"]
hits = []
for f in sorted(os.listdir(D)):
    if not f.endswith(".py"): continue
    p = D + f
    if os.path.isdir(p): continue
    try: lines = open(p, encoding="utf-8", errors="ignore").read().split("\n")
    except Exception: continue
    for i, ln in enumerate(lines, 1):
        for d in DOMS:
            if ("http://" + d) in ln:
                hits.append((f, i, d, ln.strip()[:120]))
print("残留命中 %d 处：" % len(hits))
for f, i, d, ln in hits:
    print("  %-28s L%-4d %-18s %s" % (f, i, d, ln))
if not hits: print("  （无 —— 全部已切 https）")
