prashasti
Initial changes for ai-debugger
205f6c7
Raw
History Blame Contribute Delete
4.21 kB
# Grader - evaluates a completed episode and returns a score in [0.0, 1.0].
# Scoring formula: Base score is a weighted combination of:
# - Resolution bonus (50 % weight)
# - Efficiency bonus (30 % weight) - fewer steps = higher score
# - Quality bonus (20 % weight) - average normalised reward
# All components are clipped to [0, 1] before weighting.
# Final score is also clipped to [0.0, 1.0].
from __future__ import annotations
from typing import List, Dict, Any
_RESOLUTION_WEIGHT = 0.50
_EFFICIENCY_WEIGHT = 0.30
_QUALITY_WEIGHT = 0.20
_MAX_STEPS_SIMPLE = 15
_MAX_STEPS_MULTI = 12
_MAX_STEPS_CRITICAL = 10
# Reference reward range for normalisation (empirically determined)
_REWARD_MIN = -200.0
_REWARD_MAX = 200.0
def evaluate_episode(
episode_logs: List[Dict[str, Any]],
max_steps: int = _MAX_STEPS_SIMPLE,
) -> Dict[str, Any]:
"""
Parameters
----------
episode_logs : list of step dicts containing at minimum
{"reward": float, "info": {"resolved": bool, ...}}
max_steps : maximum allowed steps for this task (used in efficiency calc)
Returns
-------
dict with keys: score (float 0-1), resolved (bool), steps (int),
total_reward (float), efficiency (float), quality (float)
"""
if not episode_logs:
return {"score": 0.0, "resolved": False, "steps": 0,
"total_reward": 0.0, "efficiency": 0.0, "quality": 0.0}
steps = len(episode_logs)
total_reward = sum(entry["reward"] for entry in episode_logs)
resolved = any(entry["info"].get("resolved", False) for entry in episode_logs)
# Component 1: resolution (binary, but partial credit for progress)
if resolved:
resolution_score = 1.0
else:
max_progress = max(
(entry["info"].get("progress", 0) for entry in episode_logs),
default=0,
)
# Guess fix sequence length ≈ 2 for all tasks
resolution_score = min(max_progress / 2.0, 0.49)
# Component 2: efficiency (resolved faster → higher score)
if resolved:
efficiency_score = max(0.0, 1.0 - (steps - 1) / max(max_steps - 1, 1))
else:
efficiency_score = 0.0
# Component 3: quality (normalised average reward)
avg_reward = total_reward / steps
norm_reward = (avg_reward - _REWARD_MIN) / (_REWARD_MAX - _REWARD_MIN)
quality_score = max(0.0, min(norm_reward, 1.0))
# Weighted combination
score = (
_RESOLUTION_WEIGHT * resolution_score +
_EFFICIENCY_WEIGHT * efficiency_score +
_QUALITY_WEIGHT * quality_score
)
score = round(max(0.0, min(score, 1.0)), 4)
return {
"score": score,
"resolved": resolved,
"steps": steps,
"total_reward": round(total_reward, 3),
"efficiency": round(efficiency_score, 4),
"quality": round(quality_score, 4),
}
if __name__ == "__main__":
import sys
import importlib
task_configs = [
("simple", "tasks.task_simple", _MAX_STEPS_SIMPLE),
("multi_service", "tasks.task_multi_service", _MAX_STEPS_MULTI),
("critical", "tasks.task_critical", _MAX_STEPS_CRITICAL),
]
from agent.baseline import act
all_passed = True
for task_name, module_path, max_steps in task_configs:
mod = importlib.import_module(module_path)
env = mod.create_env()
state = env.reset()
logs = []
done = False
while not done:
action = act(state)
next_state, reward, done, info = env.step(action)
logs.append({"reward": reward, "info": info})
state = next_state
result = evaluate_episode(logs, max_steps=max_steps)
ok = 0.0 <= result["score"] <= 1.0
all_passed = all_passed and ok
status = "PASS" if ok else "FAIL"
print(
f"[GRADER] task={task_name:<15} score={result['score']:.4f} "
f"resolved={str(result['resolved']).lower():<5} "
f"steps={result['steps']:<3} status={status}"
)
sys.exit(0 if all_passed else 1)