#!/usr/bin/env python3
"""Daily incremental crawl runner - runs all enabled crawlers and reports results."""
import json
import subprocess
import sys
import os
import time
import re

os.chdir("/root/gov_crawler")

with open("daily_crawl_config.json", "r") as f:
    cfg = json.load(f)

crawlers = [c for c in cfg["crawlers"] if c.get("enabled", False)]

results = []
total_new = 0
errors = []

def run_crawler(name, script, args_str):
    cmd_parts = []
    if script.startswith("crawl_"):
        cmd_parts = ["python3", script]
    else:
        cmd_parts = script.split()
    if args_str and args_str.strip():
        for part in args_str.split():
            cmd_parts.append(part)
    
    try:
        start = time.time()
        r = subprocess.run(cmd_parts, capture_output=True, text=True, timeout=300)
        elapsed = time.time() - start
        
        stdout = r.stdout or ""
        stderr = r.stderr or ""
        
        new_count = 0
        for line in stdout.split("\n"):
            line = line.strip()
            m = re.search(r"新[增插].*?(\d+)\s*条", line)
            if m:
                new_count += int(m.group(1))
                continue
            m = re.search(r"新[增插]入?.*?(\d+)", line)
            if m:
                new_count += int(m.group(1))
                continue
            m = re.search(r"added\s+(\d+)", line, re.I)
            if m:
                new_count += int(m.group(1))
                continue
            m = re.search(r"(\d+)\s*条新", line)
            if m:
                new_count += int(m.group(1))
                continue
        
        if r.returncode != 0:
            err_msg = stderr.strip()[:300] if stderr.strip() else "exit code " + str(r.returncode)
            return False, 0, err_msg, elapsed
        
        brief = stdout.strip()[-200:] if stdout.strip() else "(no output)"
        return True, new_count, brief, elapsed
    
    except subprocess.TimeoutExpired:
        return False, 0, "TIMEOUT (300s)", 300
    except FileNotFoundError as e:
        return False, 0, "FileNotFound: " + str(e), 0
    except Exception as e:
        return False, 0, str(e)[:300], 0

total = len(crawlers)
completed = 0

ts = time.strftime("%Y-%m-%d %H:%M:%S")
print("=== Daily Incremental Crawl Start ===")
print("Time: " + ts)
print("Total enabled sites: " + str(total))
print()

for c in crawlers:
    script = c["script"]
    name = c["name"]
    args_str = c.get("args", "")
    
    sys.stdout.write("[" + str(completed+1) + "/" + str(total) + "] " + name + "... ")
    sys.stdout.flush()
    
    success, new_count, msg, elapsed = run_crawler(name, script, args_str)
    
    results.append({
        "name": name,
        "script": script,
        "success": success,
        "new_count": new_count,
        "elapsed": round(elapsed, 1)
    })
    
    total_new += new_count
    
    if success:
        print("OK +" + str(new_count) + " [" + str(round(elapsed, 1)) + "s]")
    else:
        print("FAIL: " + msg + " [" + str(round(elapsed, 1)) + "s]")
        errors.append(name + ": " + msg)
    
    completed += 1

print()
ts2 = time.strftime("%Y-%m-%d %H:%M:%S")
print("=== CRAWL COMPLETE ===")
print("Time: " + ts2)
print("Sites: " + str(completed))
print("New items: " + str(total_new))
print("Failures: " + str(len(errors)))

if errors:
    print()
    print("Failure details:")
    for e in errors[:20]:
        print("  - " + e)
    if len(errors) > 20:
        print("  ... and " + str(len(errors)-20) + " more")

# Save structured results
with open("/tmp/daily_crawl_results.json", "w") as f:
    json.dump({
        "total_sites": completed,
        "total_new": total_new,
        "error_count": len(errors),
        "results": results,
        "errors": errors[:50],
        "timestamp": ts2
    }, f, ensure_ascii=False, indent=2)

print()
print("Results saved to /tmp/daily_crawl_results.json")
