{ "checkpoint": "models/v004-contextual", "quantization": "FP32", "platform": "Windows-11-10.0.26200-SP0", "torch_version": "2.14.0+cpu", "threads": 2, "load_ms": 11.063799960538745, "disk_bytes": 532696, "artifact_sha256": "8bff26cf3b8cb4c76f971620cb281159cfe37743d953eaf68f7eace5b1534f66", "observed_process_rss_bytes": 281284608, "memory_scope": "Observed Python RSS including training-library imports and evaluation tensors, not browser or exact peak", "end_to_end_policy_ms": { "median": 1.1031999601982534, "p95": 1.6467999666929245 }, "neural_forward_ms": { "median": 0.6952500552870333, "p95": 1.0463000508025289 }, "feature_encoding_ms": { "median": 0.4278999986127019, "p95": 0.7296999683603644 }, "python_cpu_ms": { "median": 0.0, "p95": 15.625 }, "evaluation": { "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 2.777576446533203e-05, "target_ece": 0.0003707166906679049, "candidate_recall": 1.0 }, "test": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0, "target_ece": 0.0005633262626361102, "candidate_recall": 1.0 }, "novel_wording": { "samples": 480, "action_accuracy": 0.8520833253860474, "target_accuracy": 0.9458333253860474, "joint_step_accuracy": 0.8395833373069763, "action_ece": 0.13770944606221747, "target_ece": 0.046679925406351686, "candidate_recall": 1.0 } }, "target_vps_validated": false, "scope": "Synthetic single-step benchmark. Two torch threads, not two-vCPU CPU affinity or target EPYC." }