#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""批量修复「手动写 gov_search」的 crawl_*.py。

背景：库上 AFTER INSERT 触发器已自动维护 gov_search。脚本再手动插同一 rowid →
IntegrityError → 因 commit 在两条之后，gov_raw 那条一并回滚 → 静默丢数据
（表现为「每天成功 / new=0」）。

修法：把 c.execute("INSERT ... INTO gov_search ...", (...)) 整条语句注释掉，
      保留原缩进与后续 commit。逐文件备份 + ast 校验。
"""
import ast
import io
import os
import re
import shutil
import sys

D = "/root/gov_crawler"
ARCH = os.path.join(D, "Archive")
DRY = "--dry-run" in sys.argv

pat_gs = re.compile(r"INSERT\s+(?:OR\s+REPLACE\s+)?INTO\s+gov_search", re.I)
pat_gr = re.compile(r"INSERT\s+(?:OR\s+(?:IGNORE|REPLACE)\s+)?INTO\s+gov_raw", re.I)

targets = []
for fn in sorted(os.listdir(D)):
    if not fn.startswith("crawl_") or not fn.endswith(".py"):
        continue
    src = io.open(os.path.join(D, fn), encoding="utf-8", errors="ignore").read()
    mgs = list(pat_gs.finditer(src))
    if not mgs:
        continue
    mgr = list(pat_gr.finditer(src))
    for gs in mgs:
        for gr in mgr:
            if gr.start() < gs.start() and "commit" not in src[gr.end():gs.start()] \
               and len(src[gr.end():gs.start()]) < 1200:
                targets.append(fn)
                break
        else:
            continue
        break

print("待修 crawl_*.py: %d 个" % len(targets))
print()

# 通用改法：把包含 gov_search 的那条 execute 语句整体注释掉
STMT = re.compile(
    r"^(?P<ind>[ \t]*)c(?:ur)?\.execute\(\s*[\"']INSERT\s+(?:OR\s+REPLACE\s+)?INTO\s+gov_search.*?"
    r"\)\s*$",
    re.I | re.M | re.S)

fixed, skipped = [], []
for fn in targets:
    p = os.path.join(D, fn)
    src = io.open(p, encoding="utf-8").read()
    # 逐行找：以 execute(" 开头且语句内含 gov_search 的多行调用
    lines = src.split("\n")
    out = []
    i = 0
    n_comment = 0
    while i < len(lines):
        ln = lines[i]
        if re.search(r'\.execute\(\s*["\']INSERT\s+(?:OR\s+REPLACE\s+)?INTO\s+gov_search', ln, re.I):
            # 向后收集到括号配平
            depth = ln.count("(") - ln.count(")")
            block = [ln]
            j = i
            while depth > 0 and j + 1 < len(lines):
                j += 1
                block.append(lines[j])
                depth += lines[j].count("(") - lines[j].count(")")
            ind = len(ln) - len(ln.lstrip())
            pad = " " * ind
            out.append(pad + "# 2026-09-22 修复：删除手动 gov_search 插入（库上触发器已自动维护 FTS；")
            out.append(pad + "#   手动再插同一 rowid 会 IntegrityError → 连带 gov_raw 回滚 → 静默丢数据）")
            for b in block:
                out.append("# " + b)
            n_comment += 1
            i = j + 1
            continue
        out.append(ln)
        i += 1
    new = "\n".join(out)
    if n_comment == 0:
        skipped.append(fn)
        continue
    try:
        ast.parse(new)
    except SyntaxError as e:
        print("  ❌ %s 语法失败，跳过: %s" % (fn, str(e)[:60]))
        skipped.append(fn)
        continue
    if not DRY:
        os.makedirs(ARCH, exist_ok=True)
        bak = os.path.join(ARCH, fn + ".bak_20260922_ftsdup")
        if not os.path.exists(bak):
            shutil.copy2(p, bak)
        io.open(p, "w", encoding="utf-8").write(new)
    fixed.append(fn)
    print("  ✅ %-44s 注释掉 %d 处" % (fn, n_comment))

print()
print("已修 %d 个%s" % (len(fixed), "  [DRY-RUN]" if DRY else ""))
if skipped:
    print("跳过/失败 %d 个: %s" % (len(skipped), ", ".join(skipped[:8])))
