{ "assessment": "verified", "claim": 5, "destructive_control": { "effect": "the unweighted baseline is 4-6x farther from the independently sampled fresh expectation", "mutation": "discard SOSMC importance weights and use the unweighted ImpDiff particle mean", "tracking_error_reduction_factors_from_weighting": [ 4.174582065578795, 5.547755280665158 ] }, "literal_claim": "On 2D EBM benchmark datasets, SOSMC achieves higher objective values than ImpDiff for small regularization strengths, with particle-based reward estimates that track true expectations throughout optimization (Section 5.2).", "released_native_artifact_runs": [ { "beta_kl": 0.1, "final_impdiff_fresh_reward": 0.9949566426873208, "final_sosmc_fresh_reward": 0.9985862391233444, "fresh_eval_steps": [ 0, 1000, 2000, 3000, 4000, 5000 ], "impdiff_particle_to_fresh_mae": 0.05335863728920618, "outer_iterations": 5001, "particles": 10000, "sosmc_fresh_reward_advantage": 0.0036295964360236743, "sosmc_weighted_particle_to_fresh_mae": 0.012781791434685367, "tracking_error_reduction_factor": 4.174582065578795 }, { "beta_kl": 0.25, "final_impdiff_fresh_reward": 0.9689168484330177, "final_sosmc_fresh_reward": 0.9957832426309585, "fresh_eval_steps": [ 0, 1000, 2000, 3000, 4000, 5000 ], "impdiff_particle_to_fresh_mae": 0.07853053498864175, "outer_iterations": 5001, "particles": 10000, "sosmc_fresh_reward_advantage": 0.02686639419794079, "sosmc_weighted_particle_to_fresh_mae": 0.014155371139446907, "tracking_error_reduction_factor": 5.547755280665158 } ] }