AI-debugging-agent / env /incident_generator.py
prashasti
Initial changes for ai-debugger
205f6c7
Raw
History Blame Contribute Delete
3.13 kB
"""
Incident generator - randomly samples a production incident scenario.
Each incident has a hidden root cause that the agent must diagnose
from noisy logs and degraded metrics.
"""
from __future__ import annotations
import random
from typing import Dict, List, Any
ROOT_CAUSES: Dict[str, Dict[str, Any]] = {
"api_timeout": {
"affected": ["api"],
"fix_sequence": ["scale_up", "restart_api"],
"log_hints": ["timeout error", "upstream request failed", "connection refused"],
},
"db_connection_leak": {
"affected": ["db"],
"fix_sequence": ["restart_db", "scale_up"],
"log_hints": ["too many connections", "db pool exhausted", "connection refused to db"],
},
"cache_miss_storm": {
"affected": ["cache"],
"fix_sequence": ["restart_cache", "scale_up"],
"log_hints": ["cache miss spike", "high backend load", "cache key not found"],
},
"memory_leak": {
"affected": ["api", "db"],
"fix_sequence": ["restart_api", "restart_db"],
"log_hints": ["memory usage increasing", "OOM warning", "heap allocation failure"],
},
}
_NOISE_POOL: List[str] = [
"disk warning: 78% used",
"temporary network glitch resolved",
"unrelated service restarted (metrics-exporter)",
"certificate renewal scheduled",
"cron job completed",
"health check passed for load-balancer",
"rate limiter triggered on /api/v2/bulk",
]
def generate_incident() -> Dict[str, Any]:
"""Return a fresh incident state dict."""
cause_key = random.choice(list(ROOT_CAUSES.keys()))
cause_cfg = ROOT_CAUSES[cause_key]
services = {s: "healthy" for s in ["api", "db", "cache"]}
for s in cause_cfg["affected"]:
services[s] = "degraded"
logs = _generate_logs(cause_key, cause_cfg["log_hints"])
metrics = _generate_metrics(cause_key)
return {
"services": services,
"logs": logs,
"metrics": metrics,
"root_cause": cause_key,
"fix_sequence": list(cause_cfg["fix_sequence"]), # copy
"resolved": False,
"fix_progress": 0,
}
def _generate_logs(cause: str, hints: List[str]) -> List[str]:
logs = list(hints)
# 20 % chance of a genuinely misleading entry
if random.random() < 0.2:
logs.append("corrupted log entry: [binary garbage]")
# Always add 2 noise entries
logs += random.sample(_NOISE_POOL, k=min(2, len(_NOISE_POOL)))
random.shuffle(logs)
return logs
def _generate_metrics(cause: str) -> Dict[str, float]:
base = {
"api_timeout": {"latency": 350, "error_rate": 0.55, "cpu": 75},
"db_connection_leak": {"latency": 280, "error_rate": 0.45, "cpu": 60},
"cache_miss_storm": {"latency": 220, "error_rate": 0.35, "cpu": 85},
"memory_leak": {"latency": 400, "error_rate": 0.65, "cpu": 92},
}[cause]
return {
"latency": base["latency"] + random.randint(-20, 20),
"error_rate": round(base["error_rate"] + random.uniform(-0.05, 0.05), 3),
"cpu": base["cpu"] + random.randint(-5, 5),
}