#!/usr/bin/env python3
"""Batch crawler executor - reads daily_crawl_config.json and runs all enabled crawlers"""
import json, subprocess, sys, os, re
from datetime import datetime

LOG = "/tmp/crawl_run.log"

def log(msg):
    ts = datetime.now().strftime("%H:%M:%S")
    line = f"[{ts}] {msg}"
    print(line, flush=True)
    with open(LOG, "a") as f:
        f.write(line + "\n")

with open("daily_crawl_config.json", "r") as f:
    cfg = json.load(f)

enabled = [c for c in cfg["crawlers"] if c.get("enabled", False)]
total = len(enabled)
results = []

log(f"Starting crawl of {total} enabled crawlers")

for i, c in enumerate(enabled):
    name = c["name"]
    script = c["script"]
    args = c.get("args", "").strip()
    
    if script.startswith("node "):
        cmd_parts = [s.strip() for s in script.split()]
        if args:
            cmd_parts.append(args)
    else:
        cmd_parts = ["python3", script]
        if args:
            cmd_parts.append(args)
    
    log(f"[{i+1}/{total}] {name}")
    
    try:
        r = subprocess.run(cmd_parts, capture_output=True, text=True, timeout=60, cwd="/root/gov_crawler")
        stdout = r.stdout.strip()
        stderr = r.stderr.strip()
        
        new_count = 0
        total_count = 0
        
        for line in stdout.split("\n"):
            m = re.search(r"新增[了]?(\d+)", line)
            if m: new_count += int(m.group(1))
            m = re.search(r"增量[了]?(\d+)", line)
            if m: new_count += int(m.group(1))
            m = re.search(r"new[:\s]*(\d+)", line.lower())
            if m: new_count += int(m.group(1))
            m = re.search(r"共[计有]?(\d+)", line)
            if m: total_count = int(m.group(1))
            m = re.search(r"总计[有]?(\d+)", line)
            if m: total_count = int(m.group(1))
            m = re.search(r"total[:\s]*(\d+)", line.lower())
            if m: total_count = int(m.group(1))
        
        status = "OK"
        if r.returncode != 0:
            status = f"FAIL(rc={r.returncode})"
        
        err_msg = ""
        if stderr:
            err_lines = [l for l in stderr.split("\n") if l.strip() and "warning" not in l.lower()]
            if err_lines:
                err_msg = err_lines[-1][:120]
        
        results.append((name, status, new_count, total_count, err_msg))
        log(f"  -> {status} | +{new_count} | total~{total_count}")
        
    except subprocess.TimeoutExpired:
        results.append((name, "TIMEOUT", 0, 0, ""))
        log(f"  -> TIMEOUT (60s)")
    except Exception as e:
        results.append((name, "ERROR", 0, 0, str(e)[:120]))
        log(f"  -> ERROR: {str(e)[:120]}")

# Final report
log("")
log("========== FINAL REPORT ==========")
total_new = sum(r[2] for r in results)
ok_count = sum(1 for r in results if r[1] == "OK")
fail_count = sum(1 for r in results if r[1] != "OK" and r[1] != "TIMEOUT" and not r[1].startswith("FAIL"))
fail_rc = sum(1 for r in results if r[1].startswith("FAIL"))
timeout_count = sum(1 for r in results if r[1] == "TIMEOUT")
error_count = sum(1 for r in results if r[1] == "ERROR")

log(f"Total crawlers run: {total}")
log(f"Successful: {ok_count}")
log(f"Failed (non-zero exit): {fail_rc}")
log(f"Errors: {error_count}")
log(f"Timeouts: {timeout_count}")
log(f"Total new items added (estimated): {total_new}")

issues = [r for r in results if r[1] != "OK"]
if issues:
    log(f"Sites with issues: {len(issues)}")
    for name, status, new, total_items, err in issues:
        log(f"  [{status}] {name}" + (f" - {err[:80]}" if err else ""))

log(f"Log saved to {LOG}")
print("\n=== CRAWL COMPLETE ===", flush=True)
