""" DebugOps-RX core. Simulates real-world debugging under constraints. Implements the OpenEnv spec with a probabilistic noise/drift variation engine, multi-dimensional grading, and mathematical task distribution sampling. """ import os import json import random from typing import Tuple, List, Dict from models import Action, Observation, HiddenState, ObservableState, Score # pyre-ignore BUG_TYPES = [ "logic_error", "key_error", "dependency_error", "state_corruption", "stochastic_bug" ] def sample_task_config(difficulty: str) -> dict: ranges = { "easy": {"n": (2, 4), "eta": (0.0, 0.2), "delta": (0.0, 0.1)}, "medium": {"n": (3, 8), "eta": (0.2, 0.5), "delta": (0.1, 0.3)}, "hard": {"n": (5, 12), "eta": (0.5, 0.8), "delta": (0.3, 0.6)}, "extreme":{"n": (8, 20), "eta": (0.7, 1.0), "delta": (0.5, 1.0)}, } r = ranges.get(difficulty, ranges["easy"]) return { "num_files": random.randint(*r["n"]), "bug_type": random.choice(BUG_TYPES), "eta": random.uniform(*r["eta"]), "delta": random.uniform(*r["delta"]) } def load_task_split(split: str, difficulty: str) -> dict: if split == "train": allowed = ["logic_error", "key_error"] elif split == "test": allowed = ["logic_error", "key_error"] else: # ood allowed = ["state_corruption", "stochastic_bug", "dependency_error"] while True: task = sample_task_config(difficulty) if task["bug_type"] in allowed: task["split"] = split return task def inject_noise(logs: str, eta: float) -> str: """Probabilistic corruption of logs based on eta.""" noisy = logs if random.random() < eta: noisy += "\\n[Warning] Deprecated API usage" if random.random() < 0.7 * eta: noisy = noisy.replace("service.py", "validator.py") if random.random() < eta: noisy += "\\n[Info] Latency spike detected" return noisy class DebugOpsEnv: def __init__(self, data_dir: str = "datasets", seed: int = None): self.data_dir = data_dir if seed is not None: random.seed(seed) self.state: ObservableState = None # pyre-ignore self.hidden: HiddenState = None # pyre-ignore self.trajectory: List[Action] = [] def _load_task(self, split: str, difficulty: str) -> ObservableState: # Sample configuration config = load_task_split(split, difficulty) n = config["num_files"] bug_type = config["bug_type"] eta = config["eta"] delta = config["delta"] # To build a realistic benchmark out of our 4 physical base templates, # we load the base template that closely matches the difficulty, # then mock additional files up to `n`. base_dir = f"{difficulty}_01" task_path = os.path.join(self.data_dir, base_dir) if not os.path.exists(task_path): task_path = os.path.join(self.data_dir, "easy_01") with open(os.path.join(task_path, "logs.txt")) as f: raw_logs = f.read() with open(os.path.join(task_path, "tests.txt")) as f: self._pristine_tests = f.read() # Load repo files repo_path = os.path.join(task_path, "repo") files = {} target_bug_loc = "utils.py" if os.path.exists(repo_path): for fname in os.listdir(repo_path): if fname.endswith(".py"): with open(os.path.join(repo_path, fname)) as f: files[fname] = f.read() target_bug_loc = fname # roughly heuristic # Override target logic based on physical dataset knowns if "utils.py" in files: target_bug_loc = "utils.py" elif "parser.py" in files: target_bug_loc = "parser.py" elif "service.py" in files: target_bug_loc = "service.py" elif "api.py" in files: target_bug_loc = "api.py" # Mock extra files for i in range(len(files), n): files[f"module_{i}.py"] = f"# Autogenerated mock file {i}\\ndef do_nothing():\\n pass\\n" # Hidden Truth self.hidden = HiddenState( true_bug_locations=[target_bug_loc], bug_type=bug_type, eta=eta, delta=delta, dependency_graph={} ) noisy_logs = inject_noise(raw_logs, eta) return ObservableState( files=files.copy(), original_files=files.copy(), bug_location=target_bug_loc, difficulty=difficulty, split=split, steps_taken=0, max_steps={"easy": 10, "medium": 15, "hard": 20, "extreme": 25}.get(difficulty, 15), resolved=False, files_opened=[], edits_made=[], tests_run=0, logs_analyzed=0 ) def _get_observation(self) -> Observation: return Observation( visible_files=self.state.files_opened.copy(), logs=self._current_logs if hasattr(self, "_current_logs") else "", test_results=self._current_tests if hasattr(self, "_current_tests") else None, time_remaining=self.state.max_steps - self.state.steps_taken ) def reset(self, difficulty="easy", split="test") -> Observation: self.state = self._load_task(split, difficulty) self.trajectory = [] self._current_logs = "" self._current_tests = None return self._get_observation() def _maybe_drift(self): """Temporal Drift Model.""" if random.random() < self.hidden.delta: if hasattr(self, "_current_logs") and self._current_logs: self._current_logs += "\\n[Runtime] New intermittent failure detected" def step(self, action: Action) -> Tuple[Observation, float, bool, dict]: if self.state is None: raise ValueError("Call reset() first.") reward = 0.0 done = False info = {} self.state.steps_taken += 1 self.trajectory.append(action) # 1. Action Layer if action.type == "open_file": if action.target in self.state.files: if action.target not in self.state.files_opened: self.state.files_opened.append(action.target) if action.target == self.state.bug_location: reward += 0.05 else: reward -= 0.1 elif action.type == "analyze_logs": base_dir = f"{self.state.difficulty}_01" task_path = os.path.join(self.data_dir, base_dir) if not os.path.exists(task_path): task_path = os.path.join(self.data_dir, "easy_01") with open(os.path.join(task_path, "logs.txt")) as f: base_logs = f.read() self._current_logs = inject_noise(base_logs, self.hidden.eta) self.state.logs_analyzed += 1 reward += 0.05 elif action.type == "edit_file": if action.target in self.state.files and action.content: self.state.edits_made.append({"target": action.target, "content": action.content}) self.state.files[action.target] = action.content if action.target == self.state.bug_location: reward += 0.4 else: reward -= 0.1 else: reward -= 0.1 elif action.type == "run_tests": self.state.tests_run += 1 if self._is_fixed(): self._current_tests = "✅ TESTS PASSED" reward += 1.0 self.state.resolved = True done = True else: self._current_tests = f"❌ TESTS FAILED\\n{self._pristine_tests}" reward -= 0.2 # 2. Simulate Drift self._maybe_drift() if self.state.steps_taken >= self.state.max_steps: reward -= 0.5 done = True return self._get_observation(), reward, done, info def _is_fixed(self) -> bool: diff = self.state.difficulty if not self.state.edits_made: return False latest_edits = {e["target"]: e["content"] for e in self.state.edits_made} if diff == "easy" and "utils.py" in latest_edits: if "* 10" not in latest_edits["utils.py"]: return True elif diff == "medium" and "parser.py" in latest_edits: if "valid" in latest_edits["parser.py"]: return True elif diff == "hard" and "service.py" in latest_edits: if "score" in latest_edits["service.py"] and "points" not in latest_edits["service.py"]: return True elif diff == "extreme" and "api.py" in latest_edits: if "result + 1" not in latest_edits["api.py"]: return True return False def grade(self, trajectory: List[Action]) -> Score: """Computes the final multi-dimensional vector score using strict math.""" # Correctness correctness = 1.0 if self.state.resolved else 0.0 # Efficiency efficiency = max(0.0, 1.0 - (self.state.steps_taken / self.state.max_steps)) # Reasoning (Mathematical Formula) visited = set() repeated = 0 tests = 0 seen = set() for a in trajectory: if a.type == "open_file" and a.target: visited.add(a.target) if a.type == "run_tests": tests += 1 key = (a.type, a.target) if key in seen: repeated += 1 seen.add(key) total_files = len(self.state.files) exploration = len(visited) / max(total_files, 1) test_usage = tests / max(len(trajectory), 1) redundancy = repeated / max(len(trajectory), 1) reasoning_quality = max(0.0, 0.5 * exploration + 0.3 * test_usage - 0.2 * redundancy) # Robustness wrong_edits = sum(1 for a in trajectory if a.type == "edit_file" and a.target != self.state.bug_location) robustness = max(0.0, 1.0 - (wrong_edits * 0.3)) return Score( correctness=correctness, efficiency=efficiency, reasoning_quality=reasoning_quality, robustness=robustness )