ProCreations's picture
Publish corruption-tolerant asynchronous Q-learning native reproduction
1c2b4ca verified
Raw
History Blame Contribute Delete
11.2 kB
{
"assessment": "verified",
"destructive_control_pass": true,
"minimum_vanilla_over_robust_error_ratio": 33431092.887176003,
"paired_robust_runs": [
{
"accepted_after_burn": 125228,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1299,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 0.03563053807690042,
"max_abs_q": 0.8315832572555037,
"mode": "known",
"normalized_linf_error": 0.03563053807690042,
"q_values": [
0.803655176208814,
-0.2660610990579295
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2229,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 125228,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1253,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 0.04763105869333906,
"max_abs_q": 0.8071840863200784,
"mode": "known",
"normalized_linf_error": 0.04763105869333906,
"q_values": [
0.7916546555923754,
-0.2609276133638064
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2230,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 125228,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1296,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 0.05828425209554611,
"max_abs_q": 0.7989255923514043,
"mode": "known",
"normalized_linf_error": 0.05828425209554611,
"q_values": [
0.7810014621901683,
-0.27490662818189476
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2231,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 125228,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1302,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 0.04714235852636339,
"max_abs_q": 0.8053440650180279,
"mode": "known",
"normalized_linf_error": 0.04714235852636339,
"q_values": [
0.792143355759351,
-0.2587486952553178
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2232,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 125228,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1306,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 0.045293064336337396,
"max_abs_q": 0.8142529784051116,
"mode": "known",
"normalized_linf_error": 0.045293064336337396,
"q_values": [
0.793992649949377,
-0.27432387934380625
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2233,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 125228,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1306,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 0.0496824518410611,
"max_abs_q": 0.8042019300629781,
"mode": "known",
"normalized_linf_error": 0.0496824518410611,
"q_values": [
0.8005291838293311,
-0.2818253089839182
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2234,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
}
],
"paired_vanilla_runs": [
{
"accepted_after_burn": 0,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1299,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 2227744.1273421445,
"max_abs_q": 2486070.5276853093,
"mode": "vanilla",
"normalized_linf_error": 2227744.1273421445,
"q_values": [
-2227743.2880564304,
-2175978.3726081816
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2229,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 0,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1253,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 2076587.7719096679,
"max_abs_q": 2484919.76042199,
"mode": "vanilla",
"normalized_linf_error": 2076587.7719096679,
"q_values": [
-2076586.9326239536,
-2003395.6760119009
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2230,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 0,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1296,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 1948506.2456657845,
"max_abs_q": 2596076.230117517,
"mode": "vanilla",
"normalized_linf_error": 1948506.2456657845,
"q_values": [
-1913011.5656872215,
-1948506.4778086415
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2231,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 0,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1302,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 2195869.8360337806,
"max_abs_q": 2746852.8984594396,
"mode": "vanilla",
"normalized_linf_error": 2195869.8360337806,
"q_values": [
-2195868.9967480665,
-1963006.8115141797
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2232,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 0,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1306,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 2431280.6198857073,
"max_abs_q": 2518918.812173018,
"mode": "vanilla",
"normalized_linf_error": 2431280.6198857073,
"q_values": [
-2431279.780599993,
-2201263.429200995
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2233,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
},
{
"accepted_after_burn": 0,
"alpha": 0.0004899910642875734,
"attack_magnitude": 100000000.0,
"block_gap": 1,
"burn_in": 2772,
"corruptions": 1306,
"effective_horizon": 128000,
"epsilon": 0.01,
"lambda_min": 0.3749999999999998,
"linf_error": 2200487.7265616064,
"max_abs_q": 2699155.420216447,
"mode": "vanilla",
"normalized_linf_error": 2200487.7265616064,
"q_values": [
-2200486.8872758923,
-1873704.5273021033
],
"raw_horizon": 128000,
"reward_scale": 1.0,
"seed": 2234,
"threshold_rejections": 0,
"true_q": [
0.8392857142857144,
-0.23214285714285712
],
"updates": 128000
}
],
"registered_claim": "Algorithm 1 (Robust Async-Q) combines a trimmed-mean reward estimator with an adaptive threshold G_t to filter corrupted rewards before the standard Q-learning update, requiring only knowledge of the corruption fraction ε and minimum state-action visitation probability λ_min (Algorithm 1, Assumption 1).",
"released_native_traces": [
{
"robust_sha256": "75e5e1dd96149dff28e21fb293a94d3073adfc2929fa7c47cf1893a679703180",
"robust_shape": [
3,
250000
],
"robust_tail_mean": [
0.30785562667258665,
0.3262585551498186,
0.3485561833423929
],
"vanilla_over_robust_tail_ratio": [
1838.6559164766743,
2931.073915673777,
3362.020044257801
],
"vanilla_sha256": "5ddcd482d860bd3577d218f53dd1fcbfe6cb98d52cec2109e612fa1e89bc28f4",
"vanilla_shape": [
3,
250000
],
"vanilla_tail_mean": [
566.0405694021857,
956.2879407650477,
1171.852874947122
],
"variance_label": 1
},
{
"robust_sha256": "2c63ca51f5049491eed1fab7fb2f7b47124dac3a5b1e3ca8d50199f7adfe20c2",
"robust_shape": [
3,
250000
],
"robust_tail_mean": [
0.5464117379426139,
0.5833461418975603,
0.6642962708911921
],
"vanilla_over_robust_tail_ratio": [
1032.5575210644326,
1581.4504983592396,
1760.8458798767667
],
"vanilla_sha256": "ebe98df23d3d8de71fcf7963609d6ef7e77ca1fc90bf643a4b82b7a0e0dd475b",
"vanilla_shape": [
3,
250000
],
"vanilla_tail_mean": [
564.2015496105338,
922.5330468198365,
1169.7233516162562
],
"variance_label": 5
}
],
"vanilla_over_robust_error_ratios": [
62523448.91716384,
43597346.53977085,
33431092.887176003,
46579549.78654251,
53678872.37285371,
44291045.3292679
]
}