#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""qc_delete_wide_scan.py —— 宽口径扫「清库/删数据」脚本（只读）"""
import os, re, json
D = "/root/gov_crawler/"
PAT = re.compile(r"DELETE\s+FROM\s+\w+|TRUNCATE\s+TABLE|清理旧数据|已清理|清空|删除旧数据", re.I)
live, fixed = [], []
for f in sorted(os.listdir(D)):
    if not (f.startswith("crawl_") and f.endswith(".py")): continue
    p = D + f
    if os.path.isdir(p): continue
    try: lines = open(p, encoding="utf-8", errors="ignore").read().split("\n")
    except Exception: continue
    for i, ln in enumerate(lines, 1):
        if not PAT.search(ln): continue
        s = ln.strip()
        is_comment = s.startswith("#") or "QC20260925" in ln
        rec = (f, i, s[:110])
        (fixed if is_comment else live).append(rec)
print("=== 仍在生效（未注释）的删除/清理语句：%d 处，涉及脚本 %d 个 ===" % (len(live), len(set(x[0] for x in live))))
seen = set()
for f, i, s in live:
    if f in seen: continue
    seen.add(f)
    print("  %-30s L%-4d %s" % (f, i, s))
    if len(seen) >= 28: break
print("\n=== 已被注释（今日修复）的：%d 处 / %d 个脚本（不重复列）===" % (len(fixed), len(set(x[0] for x in fixed))))
print("  " + ", ".join(sorted(set(x[0] for x in fixed))[:18]))
json.dump({"live": live, "fixed": fixed}, open(D + "qc_out/qc_delete_wide.json", "w", encoding="utf-8"), ensure_ascii=False, indent=1)
print("\n已写 qc_out/qc_delete_wide.json")
