#!/bin/bash
# 每日增量爬虫 - 服务器本地执行
# 读取 /root/gov_crawler/daily_crawl_config.json，执行所有启用的 direct_server 模式爬虫
CONFIG="/root/gov_crawler/daily_crawl_config.json"
CRAWLER_DIR="/root/gov_crawler"
LOG="/var/log/crawler_daily.log"

echo "[$(date '+%Y-%m-%d %H:%M')] 开始每日增量..." >> "$LOG"

# 读取配置，过滤启用的 direct_server 爬虫
python3 << 'PYINNER'
import json, subprocess, sys, os

CONFIG = "/root/gov_crawler/daily_crawl_config.json"
CRAWLER_DIR = "/root/gov_crawler"

cfg = json.load(open(CONFIG))
crawlers = [c for c in cfg['crawlers'] if c.get('enabled') and c.get('sync_mode') == 'direct_server']
print(f'共 {len(crawlers)} 个启用的 direct_server 爬虫')
for c in crawlers:
    script = os.path.join(CRAWLER_DIR, c['script'])
    if not os.path.isfile(script):
        print(f'  WARNING: 脚本不存在: {c["script"]}')
        continue
    try:
        cmd = ['python3', script]
        if c.get('args'):
            cmd += c['args'].split()
        to = int(c.get('timeout', 600))
        r = subprocess.run(cmd, cwd=CRAWLER_DIR, capture_output=True, text=True, timeout=to)
        out = r.stdout or ''
        err = r.stderr or ''
        summary = ''
        for line in (out + err).split('\n'):
            if any(k in line for k in ['新增', '完成', 'NEW', 'DONE', 'Error', 'ERROR']):
                summary += '  ' + line + '\n'
        if summary:
            print(f'  [{c["name"]}]\n{summary}')
        else:
            print(f'  [{c["name"]}] exit={r.returncode}')
    except subprocess.TimeoutExpired:
        print(f'  [{c["name"]}] TIMEOUT')
    except Exception as e:
        print(f'  [{c["name"]}] ERROR: {e}')
PYINNER
echo "[$(date '+%Y-%m-%d %H:%M')] 每日增量完成" >> "$LOG"
