#!/usr/bin/env python3
"""Daily incremental crawl runner - runs all enabled crawlers and reports results."""
import json
import subprocess
import sys
import os
import time
import re

os.chdir("/root/gov_crawler")

with open("daily_crawl_config.json", "r") as f:
    cfg = json.load(f)

crawlers = [c for c in cfg["crawlers"] if c.get("enabled", False)]

results = []
total_new = 0
errors = []

def run_crawler(name, script, args_str):
    cmd_parts = []
    if script.startswith("crawl_"):
        cmd_parts = ["python3", script]
    else:
        cmd_parts = script.split()
    if args_str and args_str.strip():
        for part in args_str.split():
            cmd_parts.append(part)
    
    try:
        start = time.time()
        r = subprocess.run(cmd_parts, capture_output=True, text=True, timeout=300)
        elapsed = time.time() - start
        
        stdout = r.stdout or ""
        stderr = r.stderr or ""
        
        new_count = 0
        for line in stdout.split("\n"):
            line = line.strip()
            m = re.search(r"新[增插].*?(\d+)\s*条", line)
            if m:
                new_count += int(m.group(1))
                continue
            m = re.search(r"新[增插]入?.*?(\d+)", line)
            if m:
                new_count += int(m.group(1))
                continue
            m = re.search(r"added\s+(\d+)", line, re.I)
            if m:
                new_count += int(m.group(1))
                continue
            m = re.search(r"(\d+)\s*条新", line)
            if m:
                new_count += int(m.group(1))
                continue
        
        if r.returncode != 0:
            err_msg = stderr.strip()[:300] if stderr.strip() else "exit code " + str(r.returncode)
            return False, 0, err_msg, elapsed
        
        brief = stdout.strip()[-200:] if stdout.strip() else "(no output)"
        return True, new_count, brief, elapsed
    
    except subprocess.TimeoutExpired:
        return False, 0, "TIMEOUT (300s)", 300
    except FileNotFoundError as e:
        return False, 0, "FileNotFound: " + str(e), 0
    except Exception as e:
        return False, 0, str(e)[:300], 0

total = len(crawlers)
completed = 0

ts = time.strftime("%Y-%m-%d %H:%M:%S")
print("=== Daily Incremental Crawl Start ===")
print("Time: " + ts)
print("Total enabled sites: " + str(total))
print()

for c in crawlers:
    script = c["script"]
    name = c["name"]
    args_str = c.get("args", "")
    
    sys.stdout.write("[" + str(completed+1) + "/" + str(total) + "] " + name + "... ")
    sys.stdout.flush()
    
    success, new_count, msg, elapsed = run_crawler(name, script, args_str)
    
    # 分类错误类型
    error_type = ""
    if not success:
        msg_lower = msg.lower()
        if "TIMEOUT" in msg.upper():
            error_type = "timeout"
        elif any(x in msg_lower for x in ["connection refused", "connection reset", "connect failed", "连接被拒绝"]):
            error_type = "connection_refused"
        elif any(x in msg_lower for x in ["name or service not known", "temporary failure in name", "无法解析", "dns"]):
            error_type = "dns_resolve"
        elif any(x in msg_lower for x in ["no route to host", "network is unreachable", "无法访问", "网络不可达"]):
            error_type = "unreachable"
        elif "Traceback" in msg or "traceback" in msg:
            error_type = "script_error"
        else:
            error_type = "exit_error"
    
    results.append({
        "name": name,
        "script": script,
        "success": success,
        "new_count": new_count,
        "elapsed": round(elapsed, 1),
        "error": msg if not success else "",
        "timeout": "TIMEOUT" in msg.upper(),
        "error_type": error_type
    })
    
    total_new += new_count
    
    if success:
        print("OK +" + str(new_count) + " [" + str(round(elapsed, 1)) + "s]")
    else:
        print("FAIL: " + msg + " [" + str(round(elapsed, 1)) + "s]")
        errors.append(name + ": " + msg)
    
    completed += 1

print()
ts2 = time.strftime("%Y-%m-%d %H:%M:%S")
print("=== CRAWL COMPLETE ===")
print("Time: " + ts2)
print("Sites: " + str(completed))
print("New items: " + str(total_new))
print("Failures: " + str(len(errors)))

if errors:
    print()
    print("Failure details:")
    for e in errors[:20]:
        print("  - " + e)
    if len(errors) > 20:
        print("  ... and " + str(len(errors)-20) + " more")

# Save structured results
with open("/tmp/daily_crawl_results.json", "w") as f:
    json.dump({
        "total_sites": completed,
        "total_new": total_new,
        "error_count": len(errors),
        "results": results,
        "errors": errors[:50],
        "timestamp": ts2
    }, f, ensure_ascii=False, indent=2)

print()
print("Results saved to /tmp/daily_crawl_results.json")

# 更新固化 run_status.json
ts_epoch = time.time()
status_map = {}
try:
    with open("/root/gov_crawler/run_status.json") as _f:
        status_map = json.load(_f)
except:
    pass

for r in results:
    key = r.get("script") or r.get("name", "")
    if r["success"]:
        if r.get("timeout"):
            status_str = "⏰ 超时"
        else:
            status_str = f"✅ +{r['new_count']}条"
    else:
        et = r.get("error_type", "")
        if r.get("timeout") or et == "timeout" or "TIMEOUT" in r.get("error", "").upper():
            status_str = "⏰ 超时"
        elif et in ("connection_refused", "dns_resolve", "unreachable"):
            status_str = "🚫 站点不可达"
        elif et == "script_error" or "Traceback" in r.get("error", "") or "traceback" in r.get("error", ""):
            status_str = "🐛 脚本异常"
        else:
            status_str = "❌ 失败"
    if key:
        status_map[key] = {
            "status": status_str,
            "ts": ts_epoch,
            "elapsed": r.get("elapsed", 0),
            "error_type": r.get("error_type", "")
        }

# Also try matching by name for entries that use name as key
for r in results:
    key2 = r.get("name", "")
    if key2 and key2 != r.get("script"):
        if key2 not in status_map:
            if r["success"]:
                if r.get("timeout"):
                    status_str2 = "⏰ 超时"
                else:
                    status_str2 = f"✅ +{r['new_count']}条"
            else:
                et2 = r.get("error_type", "")
                if r.get("timeout") or et2 == "timeout" or "TIMEOUT" in r.get("error", "").upper():
                    status_str2 = "⏰ 超时"
                elif et2 in ("connection_refused", "dns_resolve", "unreachable"):
                    status_str2 = "🚫 站点不可达"
                elif et2 == "script_error" or "Traceback" in r.get("error", "") or "traceback" in r.get("error", ""):
                    status_str2 = "🐛 脚本异常"
                else:
                    status_str2 = "❌ 失败"
            status_map[key2] = {
                "status": status_str2,
                "ts": ts_epoch,
                "elapsed": r.get("elapsed", 0),
                "error_type": r.get("error_type", "")
            }

with open("/root/gov_crawler/run_status.json", "w") as f:
    json.dump(status_map, f, ensure_ascii=False, indent=2)
print(f"run_status.json updated: {len(results)} entries")
