devils-agent / reports /bench-v000-mean-int8.json
devildasdf's picture
Upload experimental BAIM code, research checkpoints and measured evaluations
795f737 verified
Raw History Blame Contribute Delete
1.86 kB
{
"checkpoint": "models/v000-mean",
"quantization": "dynamic INT8 Linear only; FP32 embeddings/encoder",
"platform": "Windows-11-10.0.26200-SP0",
"torch_version": "2.14.0+cpu",
"threads": 2,
"load_ms": 24.662300013005733,
"disk_bytes": 459286,
"artifact_sha256": "4aecb5110e915d534dfa9756581cede43739692993ecef18715b800d843a1c1f",
"observed_process_rss_bytes": 295170048,
"memory_scope": "Observed Python RSS including training-library imports and evaluation tensors, not browser or exact peak",
"end_to_end_policy_ms": {
"median": 2.525599964428693,
"p95": 4.169900086708367
},
"neural_forward_ms": {
"median": 1.2204499798826873,
"p95": 2.480900031514466
},
"feature_encoding_ms": {
"median": 1.2716000201180577,
"p95": 2.3291000397875905
},
"python_cpu_ms": {
"median": 0.0,
"p95": 31.25
},
"evaluation": {
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0,
"target_ece": 1.7523765563964844e-05,
"candidate_recall": 1.0
},
"test": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0,
"target_ece": 6.67572021484375e-06,
"candidate_recall": 1.0
},
"novel_wording": {
"samples": 480,
"action_accuracy": 0.574999988079071,
"target_accuracy": 0.9229166507720947,
"joint_step_accuracy": 0.5562499761581421,
"action_ece": 0.419230160448933,
"target_ece": 0.06784084206447005,
"candidate_recall": 1.0
}
},
"target_vps_validated": false,
"scope": "Synthetic single-step benchmark. Two torch threads, not two-vCPU CPU affinity or target EPYC."
}