#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# fix_jxth_base.py — 打印 BASE/http 引用并切 https，然后实跑
import ast, os, re, shutil, sqlite3, subprocess
D = '/root/gov_crawler/'
f = 'crawl_jxth.py'
p = D + f
src = open(p, encoding='utf-8').read()
print('=== 全部 http:// 引用（含行号）===')
for i, ln in enumerate(src.split('\n'), 1):
    if 'http://' in ln or re.search(r'^\s*BASE\s*=', ln):
        print('%4d|%s' % (i, ln.rstrip()[:150]))
print()
if 'https://www.jxth.gov.cn' not in src and 'http://www.jxth.gov.cn' in src:
    src2 = src.replace('http://www.jxth.gov.cn', 'https://www.jxth.gov.cn')
    ast.parse(src2)
    b = D + 'Archive/' + f + '.bak_20260926_base'
    if not os.path.exists(b):
        shutil.copy2(p, b)
    open(p, 'w', encoding='utf-8').write(src2)
    print('OK 已切 https（备份 %s）' % os.path.basename(b))
else:
    print('无需替换或已是 https')
con = sqlite3.connect('file:/root/search.db?mode=ro', uri=True, timeout=30)
n0 = con.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE '%jxth.gov.cn%'").fetchone()[0]
print('跑前行数:', n0)
try:
    r = subprocess.run(['python3', p], capture_output=True, text=True, timeout=180, cwd=D)
    print('EXIT=%d' % r.returncode)
    for x in [y for y in (r.stdout or '').split('\n') if y.strip()][:12]:
        print('   out:', x[:150])
    for x in [y for y in (r.stderr or '').split('\n') if y.strip()][-3:]:
        print('   err:', x[:150])
except subprocess.TimeoutExpired:
    print('>180s（在抓详情 → 列表通了）')
con2 = sqlite3.connect('file:/root/search.db?mode=ro', uri=True, timeout=30)
n1 = con2.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE '%jxth.gov.cn%'").fetchone()[0]
print('跑后行数:', n1, ('OK +%d' % (n1 - n0)) if n1 > n0 else '未变')
