#!/bin/bash
# 每日增量爬虫 - 服务器本地执行
# 读取 /root/gov_crawler/daily_crawl_config.json，执行所有启用的 direct_server 模式爬虫
CONFIG="/root/gov_crawler/daily_crawl_config.json"
CRAWLER_DIR="/root/gov_crawler"
LOG="/var/log/crawler_daily.log"

echo "[$(date '+%Y-%m-%d %H:%M')] 开始每日增量..." >> "$LOG"

# 读取配置，过滤启用的 direct_server 爬虫
python3 -c "
import json, subprocess, sys, os
cfg = json.load(open('$CONFIG'))
crawlers = [c for c in cfg['crawlers'] if c.get('enabled') and c.get('sync_mode') == 'direct_server']
print(f'共 {len(crawlers)} 个启用的 direct_server 爬虫')
for c in crawlers:
    script = os.path.join('$CRAWLER_DIR', c['script'])
    if not os.path.isfile(script):
        print(f'  ⚠ 脚本不存在: {c[\"script\"]}')
        continue
    try:
        r = subprocess.run(['python3', script], cwd='$CRAWLER_DIR', capture_output=True, text=True, timeout=600)
        out = r.stdout or ''
        err = r.stderr or ''
        summary = ''
        for line in (out + err).split('\n'):
            if any(k in line for k in ['新增', '完成', '✅', '❌', 'Error']):
                summary += '  ' + line + '\n'
        if summary:
            print(f'  [{c[\"name\"]}]\\n{summary}')
        else:
            print(f'  [{c[\"name\"]}] exit={r.returncode}')
    except subprocess.TimeoutExpired:
        print(f'  [{c[\"name\"]}] ⏱ 超时')
    except Exception as e:
        print(f'  [{c[\"name\"]}] ❌ {e}')
" 2>&1 | tee -a "$LOG"

echo "[$(date '+%Y-%m-%d %H:%M')] 每日增量完成" >> "$LOG"
