{ "assessment": "verified", "destructive_control_pass": true, "minimum_vanilla_over_robust_error_ratio": 33431092.887176003, "paired_robust_runs": [ { "accepted_after_burn": 125228, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1299, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 0.03563053807690042, "max_abs_q": 0.8315832572555037, "mode": "known", "normalized_linf_error": 0.03563053807690042, "q_values": [ 0.803655176208814, -0.2660610990579295 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2229, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 125228, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1253, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 0.04763105869333906, "max_abs_q": 0.8071840863200784, "mode": "known", "normalized_linf_error": 0.04763105869333906, "q_values": [ 0.7916546555923754, -0.2609276133638064 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2230, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 125228, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1296, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 0.05828425209554611, "max_abs_q": 0.7989255923514043, "mode": "known", "normalized_linf_error": 0.05828425209554611, "q_values": [ 0.7810014621901683, -0.27490662818189476 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2231, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 125228, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1302, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 0.04714235852636339, "max_abs_q": 0.8053440650180279, "mode": "known", "normalized_linf_error": 0.04714235852636339, "q_values": [ 0.792143355759351, -0.2587486952553178 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2232, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 125228, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1306, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 0.045293064336337396, "max_abs_q": 0.8142529784051116, "mode": "known", "normalized_linf_error": 0.045293064336337396, "q_values": [ 0.793992649949377, -0.27432387934380625 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2233, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 125228, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1306, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 0.0496824518410611, "max_abs_q": 0.8042019300629781, "mode": "known", "normalized_linf_error": 0.0496824518410611, "q_values": [ 0.8005291838293311, -0.2818253089839182 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2234, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 } ], "paired_vanilla_runs": [ { "accepted_after_burn": 0, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1299, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 2227744.1273421445, "max_abs_q": 2486070.5276853093, "mode": "vanilla", "normalized_linf_error": 2227744.1273421445, "q_values": [ -2227743.2880564304, -2175978.3726081816 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2229, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 0, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1253, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 2076587.7719096679, "max_abs_q": 2484919.76042199, "mode": "vanilla", "normalized_linf_error": 2076587.7719096679, "q_values": [ -2076586.9326239536, -2003395.6760119009 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2230, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 0, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1296, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 1948506.2456657845, "max_abs_q": 2596076.230117517, "mode": "vanilla", "normalized_linf_error": 1948506.2456657845, "q_values": [ -1913011.5656872215, -1948506.4778086415 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2231, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 0, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1302, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 2195869.8360337806, "max_abs_q": 2746852.8984594396, "mode": "vanilla", "normalized_linf_error": 2195869.8360337806, "q_values": [ -2195868.9967480665, -1963006.8115141797 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2232, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 0, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1306, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 2431280.6198857073, "max_abs_q": 2518918.812173018, "mode": "vanilla", "normalized_linf_error": 2431280.6198857073, "q_values": [ -2431279.780599993, -2201263.429200995 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2233, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 }, { "accepted_after_burn": 0, "alpha": 0.0004899910642875734, "attack_magnitude": 100000000.0, "block_gap": 1, "burn_in": 2772, "corruptions": 1306, "effective_horizon": 128000, "epsilon": 0.01, "lambda_min": 0.3749999999999998, "linf_error": 2200487.7265616064, "max_abs_q": 2699155.420216447, "mode": "vanilla", "normalized_linf_error": 2200487.7265616064, "q_values": [ -2200486.8872758923, -1873704.5273021033 ], "raw_horizon": 128000, "reward_scale": 1.0, "seed": 2234, "threshold_rejections": 0, "true_q": [ 0.8392857142857144, -0.23214285714285712 ], "updates": 128000 } ], "registered_claim": "Algorithm 1 (Robust Async-Q) combines a trimmed-mean reward estimator with an adaptive threshold G_t to filter corrupted rewards before the standard Q-learning update, requiring only knowledge of the corruption fraction ε and minimum state-action visitation probability λ_min (Algorithm 1, Assumption 1).", "released_native_traces": [ { "robust_sha256": "75e5e1dd96149dff28e21fb293a94d3073adfc2929fa7c47cf1893a679703180", "robust_shape": [ 3, 250000 ], "robust_tail_mean": [ 0.30785562667258665, 0.3262585551498186, 0.3485561833423929 ], "vanilla_over_robust_tail_ratio": [ 1838.6559164766743, 2931.073915673777, 3362.020044257801 ], "vanilla_sha256": "5ddcd482d860bd3577d218f53dd1fcbfe6cb98d52cec2109e612fa1e89bc28f4", "vanilla_shape": [ 3, 250000 ], "vanilla_tail_mean": [ 566.0405694021857, 956.2879407650477, 1171.852874947122 ], "variance_label": 1 }, { "robust_sha256": "2c63ca51f5049491eed1fab7fb2f7b47124dac3a5b1e3ca8d50199f7adfe20c2", "robust_shape": [ 3, 250000 ], "robust_tail_mean": [ 0.5464117379426139, 0.5833461418975603, 0.6642962708911921 ], "vanilla_over_robust_tail_ratio": [ 1032.5575210644326, 1581.4504983592396, 1760.8458798767667 ], "vanilla_sha256": "ebe98df23d3d8de71fcf7963609d6ef7e77ca1fc90bf643a4b82b7a0e0dd475b", "vanilla_shape": [ 3, 250000 ], "vanilla_tail_mean": [ 564.2015496105338, 922.5330468198365, 1169.7233516162562 ], "variance_label": 5 } ], "vanilla_over_robust_error_ratios": [ 62523448.91716384, 43597346.53977085, 33431092.887176003, 46579549.78654251, 53678872.37285371, 44291045.3292679 ] }