anhtld commited on
Commit
1b84570
·
verified ·
1 Parent(s): d008427

Auto-sync: 2026-06-29 12:38:51 (part 2)

Browse files
results/paper_analysis.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
  "best_clean_key": "residual_k4_composemasked_grid035040045_noopbonus003",
3
- "generated_utc": "2026-06-29T16:26:29+00:00",
4
  "mechanism_gap": {
5
  "best_clean_vs_direct_same_ckpt": 0.07246376811594196,
6
  "best_clean_vs_h16": 0.0579710144927536,
@@ -1153,6 +1153,16 @@
1153
  "source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
1154
  "std_success": 0.010190374394925787
1155
  },
 
 
 
 
 
 
 
 
 
 
1156
  "residual_k4_consensus": {
1157
  "ci95_success": 0.04490086956521744,
1158
  "label": "K4 mean-by-type tangent consensus",
 
1
  {
2
  "best_clean_key": "residual_k4_composemasked_grid035040045_noopbonus003",
3
+ "generated_utc": "2026-06-29T16:37:59+00:00",
4
  "mechanism_gap": {
5
  "best_clean_vs_direct_same_ckpt": 0.07246376811594196,
6
  "best_clean_vs_h16": 0.0579710144927536,
 
1153
  "source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
1154
  "std_success": 0.010190374394925787
1155
  },
1156
+ "residual_k4_composemasked_l2comp002_grid035040045_noopbonus003": {
1157
+ "label": "K4 composed type-consensus tangents, masked, composite L2 penalty 0.02",
1158
+ "missing": true,
1159
+ "source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
1160
+ },
1161
+ "residual_k4_composemasked_l2comp005_grid035040045_noopbonus003": {
1162
+ "label": "K4 composed type-consensus tangents, masked, composite L2 penalty 0.05",
1163
+ "missing": true,
1164
+ "source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
1165
+ },
1166
  "residual_k4_consensus": {
1167
  "ci95_success": 0.04490086956521744,
1168
  "label": "K4 mean-by-type tangent consensus",
results/paper_analysis.md CHANGED
@@ -1,6 +1,6 @@
1
  # Paper Analysis
2
 
3
- Generated: `2026-06-29T16:26:29+00:00`
4
 
5
  ## Main Seed Statistics
6
 
@@ -48,6 +48,8 @@ Generated: `2026-06-29T16:26:29+00:00`
48
  | residual_k4_composemasked_grid035040045 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45 | 3 | 35.30% +/- 1.22 | +/- 3.02 | 56.91% | 0.410 | +5.57 pp |
49
  | residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45, no-op bonus 0.03 | 3 | 35.54% +/- 1.02 | +/- 2.53 | 57.02% | 0.411 | +5.80 pp |
50
  | residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, component no-op bonus 0.03 | 3 | 35.36% +/- 1.16 | +/- 2.88 | 56.98% | 0.413 | +5.62 pp |
 
 
51
  | repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | 3 | 34.32% +/- 1.35 | +/- 3.36 | 55.97% | 0.394 | +4.58 pp |
52
  | repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | 3 | 34.38% +/- 1.50 | +/- 3.73 | 56.05% | 0.394 | +4.64 pp |
53
  | repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | 3 | 34.14% +/- 1.48 | +/- 3.67 | 56.01% | 0.393 | +4.41 pp |
 
1
  # Paper Analysis
2
 
3
+ Generated: `2026-06-29T16:37:59+00:00`
4
 
5
  ## Main Seed Statistics
6
 
 
48
  | residual_k4_composemasked_grid035040045 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45 | 3 | 35.30% +/- 1.22 | +/- 3.02 | 56.91% | 0.410 | +5.57 pp |
49
  | residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45, no-op bonus 0.03 | 3 | 35.54% +/- 1.02 | +/- 2.53 | 57.02% | 0.411 | +5.80 pp |
50
  | residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, component no-op bonus 0.03 | 3 | 35.36% +/- 1.16 | +/- 2.88 | 56.98% | 0.413 | +5.62 pp |
51
+ | residual_k4_composemasked_l2comp002_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, composite L2 penalty 0.02 | 0 | missing | missing | missing | missing | missing |
52
+ | residual_k4_composemasked_l2comp005_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, composite L2 penalty 0.05 | 0 | missing | missing | missing | missing | missing |
53
  | repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | 3 | 34.32% +/- 1.35 | +/- 3.36 | 55.97% | 0.394 | +4.58 pp |
54
  | repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | 3 | 34.38% +/- 1.50 | +/- 3.73 | 56.05% | 0.394 | +4.64 pp |
55
  | repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | 3 | 34.14% +/- 1.48 | +/- 3.67 | 56.01% | 0.393 | +4.41 pp |
results/paper_table_status.json CHANGED
@@ -1315,6 +1315,44 @@
1315
  "best_config": null,
1316
  "gain_vs_h16_policy": 0.05623188405797103
1317
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1318
  {
1319
  "key": "retrieval_repair_nearmiss_k4_grid025035050_margin020",
1320
  "label": "K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
 
1315
  "best_config": null,
1316
  "gain_vs_h16_policy": 0.05623188405797103
1317
  },
1318
+ {
1319
+ "key": "retrieval_residual_k4_composemasked_l2comp002_grid035040045_noopbonus003",
1320
+ "label": "K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.02",
1321
+ "path": "h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
1322
+ "clean_deployment": "yes",
1323
+ "same_state_proposals": "no",
1324
+ "expert_proposal": "no",
1325
+ "story_role": "trust-radius penalty on composed local tangent candidates",
1326
+ "fallback_success": null,
1327
+ "pending_job": "14913944/14913955",
1328
+ "path_exists": false,
1329
+ "status": "pending",
1330
+ "success": null,
1331
+ "std_success": null,
1332
+ "completed_seeds": null,
1333
+ "num_completed": null,
1334
+ "best_config": null,
1335
+ "gain_vs_h16_policy": null
1336
+ },
1337
+ {
1338
+ "key": "retrieval_residual_k4_composemasked_l2comp005_grid035040045_noopbonus003",
1339
+ "label": "K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.05",
1340
+ "path": "h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
1341
+ "clean_deployment": "yes",
1342
+ "same_state_proposals": "no",
1343
+ "expert_proposal": "no",
1344
+ "story_role": "stronger trust-radius penalty on composed local tangent candidates",
1345
+ "fallback_success": null,
1346
+ "pending_job": "14913951/14913956",
1347
+ "path_exists": false,
1348
+ "status": "pending",
1349
+ "success": null,
1350
+ "std_success": null,
1351
+ "completed_seeds": null,
1352
+ "num_completed": null,
1353
+ "best_config": null,
1354
+ "gain_vs_h16_policy": null
1355
+ },
1356
  {
1357
  "key": "retrieval_repair_nearmiss_k4_grid025035050_margin020",
1358
  "label": "K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
results/paper_table_status.md CHANGED
@@ -72,6 +72,8 @@ Baseline h=16 policy: 29.74%
72
  | retrieval_residual_k4_composemasked_grid035040045 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20 | complete | 35.30% | +5.57 pp | yes | no | no | local tangent composition with anti-goal composite masks |
73
  | retrieval_residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20, no-op bonus 0.03 | complete | 35.54% | +5.80 pp | yes | no | no | local tangent composition with anti-goal composite masks on the current best typed prior |
74
  | retrieval_residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, component no-op bonus 0.03 | complete | 35.36% | +5.62 pp | yes | no | no | component-wise sparse prior on the masked local tangent composition chart |
 
 
75
  | retrieval_repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | complete | 34.32% | +4.58 pp | yes | no | no | deployment-clean corrective tangent transport from train near-misses back toward expert actions |
76
  | retrieval_repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | complete | 34.38% | +4.64 pp | yes | no | no | repair-tangent scale diagnostic for near-miss counterfactual geometry |
77
  | retrieval_repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | complete | 34.14% | +4.41 pp | yes | no | no | repair-tangent abstention diagnostic for near-miss counterfactual geometry |
 
72
  | retrieval_residual_k4_composemasked_grid035040045 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20 | complete | 35.30% | +5.57 pp | yes | no | no | local tangent composition with anti-goal composite masks |
73
  | retrieval_residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20, no-op bonus 0.03 | complete | 35.54% | +5.80 pp | yes | no | no | local tangent composition with anti-goal composite masks on the current best typed prior |
74
  | retrieval_residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, component no-op bonus 0.03 | complete | 35.36% | +5.62 pp | yes | no | no | component-wise sparse prior on the masked local tangent composition chart |
75
+ | retrieval_residual_k4_composemasked_l2comp002_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.02 | pending 14913944/14913955 | pending | pending | yes | no | no | trust-radius penalty on composed local tangent candidates |
76
+ | retrieval_residual_k4_composemasked_l2comp005_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.05 | pending 14913951/14913956 | pending | pending | yes | no | no | stronger trust-radius penalty on composed local tangent candidates |
77
  | retrieval_repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | complete | 34.32% | +4.58 pp | yes | no | no | deployment-clean corrective tangent transport from train near-misses back toward expert actions |
78
  | retrieval_repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | complete | 34.38% | +4.64 pp | yes | no | no | repair-tangent scale diagnostic for near-miss counterfactual geometry |
79
  | retrieval_repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | complete | 34.14% | +4.41 pp | yes | no | no | repair-tangent abstention diagnostic for near-miss counterfactual geometry |
scripts/build_paper_analysis.py CHANGED
@@ -361,6 +361,22 @@ METHODS = [
361
  "k4_composemasked_compbonus_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
362
  ),
363
  ),
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
364
  MethodSpec(
365
  key="repair_nearmiss_k4_grid025035050_margin020",
366
  label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
 
361
  "k4_composemasked_compbonus_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
362
  ),
363
  ),
364
+ MethodSpec(
365
+ key="residual_k4_composemasked_l2comp002_grid035040045_noopbonus003",
366
+ label="K4 composed type-consensus tangents, masked, composite L2 penalty 0.02",
367
+ summary_path=(
368
+ "h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_"
369
+ "k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
370
+ ),
371
+ ),
372
+ MethodSpec(
373
+ key="residual_k4_composemasked_l2comp005_grid035040045_noopbonus003",
374
+ label="K4 composed type-consensus tangents, masked, composite L2 penalty 0.05",
375
+ summary_path=(
376
+ "h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_"
377
+ "k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
378
+ ),
379
+ ),
380
  MethodSpec(
381
  key="repair_nearmiss_k4_grid025035050_margin020",
382
  label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
scripts/build_paper_table_status.py CHANGED
@@ -705,6 +705,26 @@ SPECS = [
705
  story_role="component-wise sparse prior on the masked local tangent composition chart",
706
  pending_job="14912561/14912562",
707
  ),
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
708
  ResultSpec(
709
  key="retrieval_repair_nearmiss_k4_grid025035050_margin020",
710
  label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
 
705
  story_role="component-wise sparse prior on the masked local tangent composition chart",
706
  pending_job="14912561/14912562",
707
  ),
708
+ ResultSpec(
709
+ key="retrieval_residual_k4_composemasked_l2comp002_grid035040045_noopbonus003",
710
+ label="K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.02",
711
+ path="h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
712
+ clean_deployment="yes",
713
+ same_state_proposals="no",
714
+ expert_proposal="no",
715
+ story_role="trust-radius penalty on composed local tangent candidates",
716
+ pending_job="14913944/14913955",
717
+ ),
718
+ ResultSpec(
719
+ key="retrieval_residual_k4_composemasked_l2comp005_grid035040045_noopbonus003",
720
+ label="K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.05",
721
+ path="h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
722
+ clean_deployment="yes",
723
+ same_state_proposals="no",
724
+ expert_proposal="no",
725
+ story_role="stronger trust-radius penalty on composed local tangent candidates",
726
+ pending_job="14913951/14913956",
727
+ ),
728
  ResultSpec(
729
  key="retrieval_repair_nearmiss_k4_grid025035050_margin020",
730
  label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
scripts/eval_maniskill_policy_rollout.py CHANGED
@@ -178,6 +178,13 @@ def main(argv: list[str] | None = None) -> int:
178
  help="Scale for adding a train-source reward-score advantage prior relative to the "
179
  "source residual anchor action.",
180
  )
 
 
 
 
 
 
 
181
  parser.add_argument(
182
  "--retrieval-residual-action-l2-penalty",
183
  type=float,
@@ -307,6 +314,9 @@ def main(argv: list[str] | None = None) -> int:
307
  retrieval_residual_source_advantage_bonus_scale=(
308
  args.retrieval_residual_source_advantage_bonus_scale
309
  ),
 
 
 
310
  retrieval_residual_action_l2_penalty=args.retrieval_residual_action_l2_penalty,
311
  retrieval_residual_scale=args.retrieval_residual_scale,
312
  retrieval_residual_scales=retrieval_residual_scales,
 
178
  help="Scale for adding a train-source reward-score advantage prior relative to the "
179
  "source residual anchor action.",
180
  )
181
+ parser.add_argument(
182
+ "--retrieval-residual-composite-l2-penalty-scale",
183
+ type=float,
184
+ default=0.0,
185
+ help="Penalty scale subtracted from composed residual candidates according to "
186
+ "their mean squared residual energy.",
187
+ )
188
  parser.add_argument(
189
  "--retrieval-residual-action-l2-penalty",
190
  type=float,
 
314
  retrieval_residual_source_advantage_bonus_scale=(
315
  args.retrieval_residual_source_advantage_bonus_scale
316
  ),
317
+ retrieval_residual_composite_l2_penalty_scale=(
318
+ args.retrieval_residual_composite_l2_penalty_scale
319
+ ),
320
  retrieval_residual_action_l2_penalty=args.retrieval_residual_action_l2_penalty,
321
  retrieval_residual_scale=args.retrieval_residual_scale,
322
  retrieval_residual_scales=retrieval_residual_scales,
scripts/slurm/eval_maniskill_policy_rollout.sbatch CHANGED
@@ -60,6 +60,7 @@ RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTAGE="${RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTA
60
  RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
61
  RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
62
  RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
 
63
  RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
64
  RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
65
  RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
@@ -138,6 +139,7 @@ apptainer exec --nv \
138
  --retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
139
  --retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
140
  --retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
 
141
  --retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
142
  --retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
143
  --retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
 
60
  RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
61
  RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
62
  RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
63
+ RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE="${RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE:-0.0}"
64
  RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
65
  RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
66
  RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
 
139
  --retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
140
  --retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
141
  --retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
142
+ --retrieval-residual-composite-l2-penalty-scale "$RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE" \
143
  --retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
144
  --retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
145
  --retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
scripts/slurm/eval_maniskill_policy_rollout_cpu_smoke.sbatch CHANGED
@@ -59,6 +59,7 @@ RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTAGE="${RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTA
59
  RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
60
  RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
61
  RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
 
62
  RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
63
  RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
64
  RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
@@ -134,6 +135,7 @@ apptainer exec \
134
  --retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
135
  --retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
136
  --retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
 
137
  --retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
138
  --retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
139
  --retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
 
59
  RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
60
  RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
61
  RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
62
+ RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE="${RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE:-0.0}"
63
  RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
64
  RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
65
  RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
 
135
  --retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
136
  --retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
137
  --retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
138
+ --retrieval-residual-composite-l2-penalty-scale "$RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE" \
139
  --retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
140
  --retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
141
  --retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
scripts/slurm/summarize_h16_policy_ckpt.sbatch CHANGED
@@ -91,6 +91,9 @@ for result_path in sorted(base_dir.glob(f"seed_*/{out_name}")):
91
  "retrieval_residual_source_advantage_bonus_scale": data.get(
92
  "retrieval_residual_source_advantage_bonus_scale", 0.0
93
  ),
 
 
 
94
  "retrieval_residual_scale": data.get("retrieval_residual_scale", 0.0),
95
  "retrieval_residual_scales": data.get("retrieval_residual_scales", []),
96
  "retrieval_residual_anchor": data.get("retrieval_residual_anchor", "none"),
 
91
  "retrieval_residual_source_advantage_bonus_scale": data.get(
92
  "retrieval_residual_source_advantage_bonus_scale", 0.0
93
  ),
94
+ "retrieval_residual_composite_l2_penalty_scale": data.get(
95
+ "retrieval_residual_composite_l2_penalty_scale", 0.0
96
+ ),
97
  "retrieval_residual_scale": data.get("retrieval_residual_scale", 0.0),
98
  "retrieval_residual_scales": data.get("retrieval_residual_scales", []),
99
  "retrieval_residual_anchor": data.get("retrieval_residual_anchor", "none"),
tests/test_maniskill_policy_rollout.py CHANGED
@@ -1022,6 +1022,35 @@ def test_retrieval_residual_reducer_builds_composed_type_tangents() -> None:
1022
  )
1023
 
1024
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1025
  def test_retrieval_residual_reducer_penalizes_low_consensus_tangents() -> None:
1026
  residuals, candidate_types, bonuses = _reduce_residual_candidates_by_type(
1027
  [
 
1022
  )
1023
 
1024
 
1025
+ def test_retrieval_residual_reducer_penalizes_composed_tangent_energy() -> None:
1026
+ residuals, candidate_types, bonuses = _reduce_residual_candidates_by_type(
1027
+ [
1028
+ [[0.0, 0.0]],
1029
+ [[0.4, 0.2]],
1030
+ [[-0.2, 0.0]],
1031
+ ],
1032
+ [
1033
+ "policy_residual",
1034
+ "residual_no_op",
1035
+ "residual_wrong_gripper",
1036
+ ],
1037
+ mode="compose_mean_by_type",
1038
+ bonuses=[0.0, 0.04, 0.02],
1039
+ composite_l2_penalty_scale=0.5,
1040
+ )
1041
+
1042
+ assert candidate_types == [
1043
+ "policy_residual",
1044
+ "residual_no_op",
1045
+ "residual_wrong_gripper",
1046
+ "residual_no_op+residual_wrong_gripper",
1047
+ ]
1048
+ assert np.allclose(np.asarray(residuals[-1], dtype=np.float32), [[0.2, 0.2]])
1049
+ # Base composite bonus is mean(0.04, 0.02)=0.03; residual energy is
1050
+ # mean([0.2^2, 0.2^2])=0.04, so a 0.5 penalty subtracts 0.02.
1051
+ assert np.allclose(bonuses, [0.0, 0.04, 0.02, 0.01])
1052
+
1053
+
1054
  def test_retrieval_residual_reducer_penalizes_low_consensus_tangents() -> None:
1055
  residuals, candidate_types, bonuses = _reduce_residual_candidates_by_type(
1056
  [