{ "model": "fusion-embedding-2-tactus", "version": "v0.1-preview", "benchmark": "STAG classification (Nature 2019), 27 objects, unseen test recordings, open-vocabulary text queries", "released_checkpoint_eval": { "top1_27way": 0.817, "top3_27way": 0.9507, "kway5": 0.948, "recording_top1": 0.7407 }, "recipe_mean_over_4_runs": { "top1_27way": 0.771, "top1_std": 0.062, "top3_27way": 0.935, "kway5": 0.935, "recording_top1": 0.722, "runs_top1": [ 0.8293, 0.7014, 0.7367, 0.817 ] }, "baselines": { "scratch_mean_n3": 0.705, "stag_2019_supervised_cnn_top1": 0.76, "chance_top1": 0.037, "chance_top3": 0.1111, "chance_kway5": 0.2 }, "note": "Evaluation mirrors STAG's cluster-sampling test protocol, not their byte-identical harness. The recipe mean exceeds the 2019 CNN by less than one standard error; we describe the result as matching to exceeding the original baseline. Run-to-run standard deviation of 0.062 is a known limitation." }