#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fix_heshan_timeout.py —— 给 config 加 timeout:900（原用默认 600，8 页跑不完被撞顶杀掉）"""
import json, os, shutil, datetime
D = "/root/gov_crawler/"; CFG = D + "daily_crawl_config.json"
txt = open(CFG, encoding="utf-8").read()
i = txt.find('"crawl_heshan_gkmlpt.py"')
if i < 0: raise SystemExit("未找到条目")
s0 = txt.rfind("{", 0, i); s1 = txt.find('"script"', i + 30)
if s1 < 0: s1 = len(txt)
blk = txt[s0:s1]
print("原条目:", blk.replace("\n", " ")[:220])
if '"timeout"' in blk:
    print("⚠️ 已有 timeout，不动")
else:
    # 在 "incremental": true, 后插入
    import re as _re
    m = _re.search(r'("incremental"\s*:\s*true\s*,)', blk)
    if not m:
        print("⚠️ 没找到 incremental 锚点，不动"); raise SystemExit
    nb = blk[:m.end()] + ' "timeout": 900,' + blk[m.end():]
    new = txt[:s0] + nb + txt[s1:]
    json.loads(new)
    bak = D + "Archive/daily_crawl_config.json.bak_%s_heshan_timeout" % datetime.date.today().strftime("%Y%m%d")
    if not os.path.exists(bak): shutil.copy2(CFG, bak)
    open(CFG, "w", encoding="utf-8").write(new)
    print("✅ 已加 timeout:900（备份 %s）" % os.path.basename(bak))
cfg = json.load(open(CFG, encoding="utf-8"))
lst = cfg if isinstance(cfg, list) else cfg.get("tasks", [])
e = [c for c in lst if c.get("script") == "crawl_heshan_gkmlpt.py"]
print("回读:", json.dumps(e[0], ensure_ascii=False) if e else "无")
print("\n=== 抓取函数原文（准备加重试）===")
src = open(D + "crawl_heshan_gkmlpt.py", encoding="utf-8").read().split("\n")
for i in range(19, 45):
    if i < len(src) and src[i].strip(): print("  L%-4d %s" % (i + 1, src[i][:112]))
