#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fix_heshan_args.py —— 给 crawl_heshan_gkmlpt.py 补 config 参数（脚本必需）"""
import json, os, re, shutil, sqlite3, subprocess, datetime
D = "/root/gov_crawler/"
CFG = D + "daily_crawl_config.json"
TARGET = "crawl_heshan_gkmlpt.py"
ARG = "龙口镇"   # 与 config name「江门鹤山市龙口镇人民政府」对应
txt = open(CFG, encoding="utf-8").read()
i = txt.find('"crawl_heshan_gkmlpt.py"')
print("找到脚本位置:", i)
if i < 0: raise SystemExit("未找到")
# 该条目的边界：往前找最近的 {，往后找下一个 "script":
s0 = txt.rfind("{", 0, i)
s1 = txt.find('"script"', i + 30)
if s1 < 0: s1 = len(txt)
block = txt[s0:s1]
print("原条目片段:", re.sub(r"\s+", " ", block)[:300])
m = re.search(r'"args"\s*:\s*(""|null|\[[^\]]*\])', block)
if not m:
    print("⚠️ 该条目里没有 args 键，需手工加")
else:
    if m.group(1) == '""':
        nb = block[:m.start()] + '"args": "%s"' % ARG + block[m.end():]
        new = txt[:s0] + nb + txt[s1:]
        json.loads(new)   # 校验
        bak = D + "Archive/daily_crawl_config.json.bak_%s_heshan" % datetime.date.today().strftime("%Y%m%d")
        if not os.path.exists(bak): shutil.copy2(CFG, bak)
        open(CFG, "w", encoding="utf-8").write(new)
        print("✅ 已写入 args=%s（备份 %s）" % (ARG, os.path.basename(bak)))
    else:
        print("args 原值非空:", m.group(1), "→ 不动")
cfg = json.load(open(CFG, encoding="utf-8"))
ent = [c for c in (cfg if isinstance(cfg, list) else cfg.get("tasks", [])) if c.get("script") == TARGET]
print("回读:", json.dumps(ent[0], ensure_ascii=False) if ent else "无")
db = sqlite3.connect("file:/root/search.db?mode=ro", uri=True, timeout=30)
before = db.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE '%hbdawu%' OR page_url LIKE '%heshan%' OR site_name LIKE '%鹤山%'").fetchone()[0]
print("跑前相关行数:", before)
r = subprocess.run(["python3", D + TARGET, ARG], capture_output=True, text=True, timeout=110, cwd=D)
lines = [x for x in (r.stdout or "").strip().split("\n") if x.strip()][-3:]
print("实跑 EXIT=%d | %s" % (r.returncode, " / ".join(x[:70] for x in lines)))
db2 = sqlite3.connect("file:/root/search.db?mode=ro", uri=True, timeout=30)
after = db2.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE '%hbdawu%' OR page_url LIKE '%heshan%' OR site_name LIKE '%鹤山%'").fetchone()[0]
print("跑后相关行数: %d (%s)" % (after, "✅增加" if after > before else "未变"))
