devils-agent / reports /bench-v002-gru-fp32.json
devildasdf's picture
Upload experimental BAIM code, research checkpoints and measured evaluations
795f737 verified
Raw History Blame Contribute Delete
1.81 kB
{
"checkpoint": "models/v002-gru",
"quantization": "FP32",
"platform": "Windows-11-10.0.26200-SP0",
"torch_version": "2.14.0+cpu",
"threads": 2,
"load_ms": 66.8565999949351,
"disk_bytes": 616488,
"artifact_sha256": "1a8b60b1fa35627c5cd5a5791b9d160e337e99c4d6ffa5c4c3015432f68fb54e",
"observed_process_rss_bytes": 297365504,
"memory_scope": "Observed Python RSS including training-library imports and evaluation tensors, not browser or exact peak",
"end_to_end_policy_ms": {
"median": 4.623400047421455,
"p95": 6.873500067740679
},
"neural_forward_ms": {
"median": 3.207450034096837,
"p95": 4.629200091585517
},
"feature_encoding_ms": {
"median": 1.2757499353028834,
"p95": 2.2782000014558434
},
"python_cpu_ms": {
"median": 0.0,
"p95": 31.25
},
"evaluation": {
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0,
"target_ece": 0.0002561807632446289,
"candidate_recall": 1.0
},
"test": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0,
"target_ece": 0.0005216506760916673,
"candidate_recall": 1.0
},
"novel_wording": {
"samples": 480,
"action_accuracy": 0.3812499940395355,
"target_accuracy": 0.4312500059604645,
"joint_step_accuracy": 0.3583333194255829,
"action_ece": 0.6094635868794285,
"target_ece": 0.5180023244465701,
"candidate_recall": 1.0
}
},
"target_vps_validated": false,
"scope": "Synthetic single-step benchmark. Two torch threads, not two-vCPU CPU affinity or target EPYC."
}