Auto-sync: 2026-06-29 12:38:51 (part 2)
Browse files- results/paper_analysis.json +11 -1
- results/paper_analysis.md +3 -1
- results/paper_table_status.json +38 -0
- results/paper_table_status.md +2 -0
- scripts/build_paper_analysis.py +16 -0
- scripts/build_paper_table_status.py +20 -0
- scripts/eval_maniskill_policy_rollout.py +10 -0
- scripts/slurm/eval_maniskill_policy_rollout.sbatch +2 -0
- scripts/slurm/eval_maniskill_policy_rollout_cpu_smoke.sbatch +2 -0
- scripts/slurm/summarize_h16_policy_ckpt.sbatch +3 -0
- tests/test_maniskill_policy_rollout.py +29 -0
results/paper_analysis.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"best_clean_key": "residual_k4_composemasked_grid035040045_noopbonus003",
|
| 3 |
-
"generated_utc": "2026-06-29T16:
|
| 4 |
"mechanism_gap": {
|
| 5 |
"best_clean_vs_direct_same_ckpt": 0.07246376811594196,
|
| 6 |
"best_clean_vs_h16": 0.0579710144927536,
|
|
@@ -1153,6 +1153,16 @@
|
|
| 1153 |
"source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
|
| 1154 |
"std_success": 0.010190374394925787
|
| 1155 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1156 |
"residual_k4_consensus": {
|
| 1157 |
"ci95_success": 0.04490086956521744,
|
| 1158 |
"label": "K4 mean-by-type tangent consensus",
|
|
|
|
| 1 |
{
|
| 2 |
"best_clean_key": "residual_k4_composemasked_grid035040045_noopbonus003",
|
| 3 |
+
"generated_utc": "2026-06-29T16:37:59+00:00",
|
| 4 |
"mechanism_gap": {
|
| 5 |
"best_clean_vs_direct_same_ckpt": 0.07246376811594196,
|
| 6 |
"best_clean_vs_h16": 0.0579710144927536,
|
|
|
|
| 1153 |
"source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
|
| 1154 |
"std_success": 0.010190374394925787
|
| 1155 |
},
|
| 1156 |
+
"residual_k4_composemasked_l2comp002_grid035040045_noopbonus003": {
|
| 1157 |
+
"label": "K4 composed type-consensus tangents, masked, composite L2 penalty 0.02",
|
| 1158 |
+
"missing": true,
|
| 1159 |
+
"source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
|
| 1160 |
+
},
|
| 1161 |
+
"residual_k4_composemasked_l2comp005_grid035040045_noopbonus003": {
|
| 1162 |
+
"label": "K4 composed type-consensus tangents, masked, composite L2 penalty 0.05",
|
| 1163 |
+
"missing": true,
|
| 1164 |
+
"source": "results/h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
|
| 1165 |
+
},
|
| 1166 |
"residual_k4_consensus": {
|
| 1167 |
"ci95_success": 0.04490086956521744,
|
| 1168 |
"label": "K4 mean-by-type tangent consensus",
|
results/paper_analysis.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
# Paper Analysis
|
| 2 |
|
| 3 |
-
Generated: `2026-06-29T16:
|
| 4 |
|
| 5 |
## Main Seed Statistics
|
| 6 |
|
|
@@ -48,6 +48,8 @@ Generated: `2026-06-29T16:26:29+00:00`
|
|
| 48 |
| residual_k4_composemasked_grid035040045 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45 | 3 | 35.30% +/- 1.22 | +/- 3.02 | 56.91% | 0.410 | +5.57 pp |
|
| 49 |
| residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45, no-op bonus 0.03 | 3 | 35.54% +/- 1.02 | +/- 2.53 | 57.02% | 0.411 | +5.80 pp |
|
| 50 |
| residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, component no-op bonus 0.03 | 3 | 35.36% +/- 1.16 | +/- 2.88 | 56.98% | 0.413 | +5.62 pp |
|
|
|
|
|
|
|
| 51 |
| repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | 3 | 34.32% +/- 1.35 | +/- 3.36 | 55.97% | 0.394 | +4.58 pp |
|
| 52 |
| repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | 3 | 34.38% +/- 1.50 | +/- 3.73 | 56.05% | 0.394 | +4.64 pp |
|
| 53 |
| repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | 3 | 34.14% +/- 1.48 | +/- 3.67 | 56.01% | 0.393 | +4.41 pp |
|
|
|
|
| 1 |
# Paper Analysis
|
| 2 |
|
| 3 |
+
Generated: `2026-06-29T16:37:59+00:00`
|
| 4 |
|
| 5 |
## Main Seed Statistics
|
| 6 |
|
|
|
|
| 48 |
| residual_k4_composemasked_grid035040045 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45 | 3 | 35.30% +/- 1.22 | +/- 3.02 | 56.91% | 0.410 | +5.57 pp |
|
| 49 |
| residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, scales 0.35/0.40/0.45, no-op bonus 0.03 | 3 | 35.54% +/- 1.02 | +/- 2.53 | 57.02% | 0.411 | +5.80 pp |
|
| 50 |
| residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, component no-op bonus 0.03 | 3 | 35.36% +/- 1.16 | +/- 2.88 | 56.98% | 0.413 | +5.62 pp |
|
| 51 |
+
| residual_k4_composemasked_l2comp002_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, composite L2 penalty 0.02 | 0 | missing | missing | missing | missing | missing |
|
| 52 |
+
| residual_k4_composemasked_l2comp005_grid035040045_noopbonus003 | K4 composed type-consensus tangents, masked, composite L2 penalty 0.05 | 0 | missing | missing | missing | missing | missing |
|
| 53 |
| repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | 3 | 34.32% +/- 1.35 | +/- 3.36 | 55.97% | 0.394 | +4.58 pp |
|
| 54 |
| repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | 3 | 34.38% +/- 1.50 | +/- 3.73 | 56.05% | 0.394 | +4.64 pp |
|
| 55 |
| repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | 3 | 34.14% +/- 1.48 | +/- 3.67 | 56.01% | 0.393 | +4.41 pp |
|
results/paper_table_status.json
CHANGED
|
@@ -1315,6 +1315,44 @@
|
|
| 1315 |
"best_config": null,
|
| 1316 |
"gain_vs_h16_policy": 0.05623188405797103
|
| 1317 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1318 |
{
|
| 1319 |
"key": "retrieval_repair_nearmiss_k4_grid025035050_margin020",
|
| 1320 |
"label": "K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
|
|
|
|
| 1315 |
"best_config": null,
|
| 1316 |
"gain_vs_h16_policy": 0.05623188405797103
|
| 1317 |
},
|
| 1318 |
+
{
|
| 1319 |
+
"key": "retrieval_residual_k4_composemasked_l2comp002_grid035040045_noopbonus003",
|
| 1320 |
+
"label": "K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.02",
|
| 1321 |
+
"path": "h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
|
| 1322 |
+
"clean_deployment": "yes",
|
| 1323 |
+
"same_state_proposals": "no",
|
| 1324 |
+
"expert_proposal": "no",
|
| 1325 |
+
"story_role": "trust-radius penalty on composed local tangent candidates",
|
| 1326 |
+
"fallback_success": null,
|
| 1327 |
+
"pending_job": "14913944/14913955",
|
| 1328 |
+
"path_exists": false,
|
| 1329 |
+
"status": "pending",
|
| 1330 |
+
"success": null,
|
| 1331 |
+
"std_success": null,
|
| 1332 |
+
"completed_seeds": null,
|
| 1333 |
+
"num_completed": null,
|
| 1334 |
+
"best_config": null,
|
| 1335 |
+
"gain_vs_h16_policy": null
|
| 1336 |
+
},
|
| 1337 |
+
{
|
| 1338 |
+
"key": "retrieval_residual_k4_composemasked_l2comp005_grid035040045_noopbonus003",
|
| 1339 |
+
"label": "K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.05",
|
| 1340 |
+
"path": "h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
|
| 1341 |
+
"clean_deployment": "yes",
|
| 1342 |
+
"same_state_proposals": "no",
|
| 1343 |
+
"expert_proposal": "no",
|
| 1344 |
+
"story_role": "stronger trust-radius penalty on composed local tangent candidates",
|
| 1345 |
+
"fallback_success": null,
|
| 1346 |
+
"pending_job": "14913951/14913956",
|
| 1347 |
+
"path_exists": false,
|
| 1348 |
+
"status": "pending",
|
| 1349 |
+
"success": null,
|
| 1350 |
+
"std_success": null,
|
| 1351 |
+
"completed_seeds": null,
|
| 1352 |
+
"num_completed": null,
|
| 1353 |
+
"best_config": null,
|
| 1354 |
+
"gain_vs_h16_policy": null
|
| 1355 |
+
},
|
| 1356 |
{
|
| 1357 |
"key": "retrieval_repair_nearmiss_k4_grid025035050_margin020",
|
| 1358 |
"label": "K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
|
results/paper_table_status.md
CHANGED
|
@@ -72,6 +72,8 @@ Baseline h=16 policy: 29.74%
|
|
| 72 |
| retrieval_residual_k4_composemasked_grid035040045 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20 | complete | 35.30% | +5.57 pp | yes | no | no | local tangent composition with anti-goal composite masks |
|
| 73 |
| retrieval_residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20, no-op bonus 0.03 | complete | 35.54% | +5.80 pp | yes | no | no | local tangent composition with anti-goal composite masks on the current best typed prior |
|
| 74 |
| retrieval_residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, component no-op bonus 0.03 | complete | 35.36% | +5.62 pp | yes | no | no | component-wise sparse prior on the masked local tangent composition chart |
|
|
|
|
|
|
|
| 75 |
| retrieval_repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | complete | 34.32% | +4.58 pp | yes | no | no | deployment-clean corrective tangent transport from train near-misses back toward expert actions |
|
| 76 |
| retrieval_repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | complete | 34.38% | +4.64 pp | yes | no | no | repair-tangent scale diagnostic for near-miss counterfactual geometry |
|
| 77 |
| retrieval_repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | complete | 34.14% | +4.41 pp | yes | no | no | repair-tangent abstention diagnostic for near-miss counterfactual geometry |
|
|
|
|
| 72 |
| retrieval_residual_k4_composemasked_grid035040045 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20 | complete | 35.30% | +5.57 pp | yes | no | no | local tangent composition with anti-goal composite masks |
|
| 73 |
| retrieval_residual_k4_composemasked_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, scales 0.35/0.40/0.45, margin 0.20, no-op bonus 0.03 | complete | 35.54% | +5.80 pp | yes | no | no | local tangent composition with anti-goal composite masks on the current best typed prior |
|
| 74 |
| retrieval_residual_k4_composemasked_compbonus_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, component no-op bonus 0.03 | complete | 35.36% | +5.62 pp | yes | no | no | component-wise sparse prior on the masked local tangent composition chart |
|
| 75 |
+
| retrieval_residual_k4_composemasked_l2comp002_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.02 | pending 14913944/14913955 | pending | pending | yes | no | no | trust-radius penalty on composed local tangent candidates |
|
| 76 |
+
| retrieval_residual_k4_composemasked_l2comp005_grid035040045_noopbonus003 | K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.05 | pending 14913951/14913956 | pending | pending | yes | no | no | stronger trust-radius penalty on composed local tangent candidates |
|
| 77 |
| retrieval_repair_nearmiss_k4_grid025035050_margin020 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20 | complete | 34.32% | +4.58 pp | yes | no | no | deployment-clean corrective tangent transport from train near-misses back toward expert actions |
|
| 78 |
| retrieval_repair_nearmiss_k4_grid035050075_margin020 | K4 near-miss-to-expert repair tangent, scales 0.35/0.50/0.75, margin 0.20 | complete | 34.38% | +4.64 pp | yes | no | no | repair-tangent scale diagnostic for near-miss counterfactual geometry |
|
| 79 |
| retrieval_repair_nearmiss_k4_grid025035050_margin010 | K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.10 | complete | 34.14% | +4.41 pp | yes | no | no | repair-tangent abstention diagnostic for near-miss counterfactual geometry |
|
scripts/build_paper_analysis.py
CHANGED
|
@@ -361,6 +361,22 @@ METHODS = [
|
|
| 361 |
"k4_composemasked_compbonus_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
|
| 362 |
),
|
| 363 |
),
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 364 |
MethodSpec(
|
| 365 |
key="repair_nearmiss_k4_grid025035050_margin020",
|
| 366 |
label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
|
|
|
|
| 361 |
"k4_composemasked_compbonus_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
|
| 362 |
),
|
| 363 |
),
|
| 364 |
+
MethodSpec(
|
| 365 |
+
key="residual_k4_composemasked_l2comp002_grid035040045_noopbonus003",
|
| 366 |
+
label="K4 composed type-consensus tangents, masked, composite L2 penalty 0.02",
|
| 367 |
+
summary_path=(
|
| 368 |
+
"h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_"
|
| 369 |
+
"k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
|
| 370 |
+
),
|
| 371 |
+
),
|
| 372 |
+
MethodSpec(
|
| 373 |
+
key="residual_k4_composemasked_l2comp005_grid035040045_noopbonus003",
|
| 374 |
+
label="K4 composed type-consensus tangents, masked, composite L2 penalty 0.05",
|
| 375 |
+
summary_path=(
|
| 376 |
+
"h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_"
|
| 377 |
+
"k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json"
|
| 378 |
+
),
|
| 379 |
+
),
|
| 380 |
MethodSpec(
|
| 381 |
key="repair_nearmiss_k4_grid025035050_margin020",
|
| 382 |
label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
|
scripts/build_paper_table_status.py
CHANGED
|
@@ -705,6 +705,26 @@ SPECS = [
|
|
| 705 |
story_role="component-wise sparse prior on the masked local tangent composition chart",
|
| 706 |
pending_job="14912561/14912562",
|
| 707 |
),
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 708 |
ResultSpec(
|
| 709 |
key="retrieval_repair_nearmiss_k4_grid025035050_margin020",
|
| 710 |
label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
|
|
|
|
| 705 |
story_role="component-wise sparse prior on the masked local tangent composition chart",
|
| 706 |
pending_job="14912561/14912562",
|
| 707 |
),
|
| 708 |
+
ResultSpec(
|
| 709 |
+
key="retrieval_residual_k4_composemasked_l2comp002_grid035040045_noopbonus003",
|
| 710 |
+
label="K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.02",
|
| 711 |
+
path="h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp002_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
|
| 712 |
+
clean_deployment="yes",
|
| 713 |
+
same_state_proposals="no",
|
| 714 |
+
expert_proposal="no",
|
| 715 |
+
story_role="trust-radius penalty on composed local tangent candidates",
|
| 716 |
+
pending_job="14913944/14913955",
|
| 717 |
+
),
|
| 718 |
+
ResultSpec(
|
| 719 |
+
key="retrieval_residual_k4_composemasked_l2comp005_grid035040045_noopbonus003",
|
| 720 |
+
label="K4 composed type-consensus residual retrieval, masked, composite L2 penalty 0.05",
|
| 721 |
+
path="h16_policy_ckpt_near_miss_policy_bc5_bestpt_retrieval_residual_k4_composemasked_l2comp005_grid035040045_safe_margin0p20_noopbonus0p03_summary.json",
|
| 722 |
+
clean_deployment="yes",
|
| 723 |
+
same_state_proposals="no",
|
| 724 |
+
expert_proposal="no",
|
| 725 |
+
story_role="stronger trust-radius penalty on composed local tangent candidates",
|
| 726 |
+
pending_job="14913951/14913956",
|
| 727 |
+
),
|
| 728 |
ResultSpec(
|
| 729 |
key="retrieval_repair_nearmiss_k4_grid025035050_margin020",
|
| 730 |
label="K4 near-miss-to-expert repair tangent, scales 0.25/0.35/0.50, margin 0.20",
|
scripts/eval_maniskill_policy_rollout.py
CHANGED
|
@@ -178,6 +178,13 @@ def main(argv: list[str] | None = None) -> int:
|
|
| 178 |
help="Scale for adding a train-source reward-score advantage prior relative to the "
|
| 179 |
"source residual anchor action.",
|
| 180 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 181 |
parser.add_argument(
|
| 182 |
"--retrieval-residual-action-l2-penalty",
|
| 183 |
type=float,
|
|
@@ -307,6 +314,9 @@ def main(argv: list[str] | None = None) -> int:
|
|
| 307 |
retrieval_residual_source_advantage_bonus_scale=(
|
| 308 |
args.retrieval_residual_source_advantage_bonus_scale
|
| 309 |
),
|
|
|
|
|
|
|
|
|
|
| 310 |
retrieval_residual_action_l2_penalty=args.retrieval_residual_action_l2_penalty,
|
| 311 |
retrieval_residual_scale=args.retrieval_residual_scale,
|
| 312 |
retrieval_residual_scales=retrieval_residual_scales,
|
|
|
|
| 178 |
help="Scale for adding a train-source reward-score advantage prior relative to the "
|
| 179 |
"source residual anchor action.",
|
| 180 |
)
|
| 181 |
+
parser.add_argument(
|
| 182 |
+
"--retrieval-residual-composite-l2-penalty-scale",
|
| 183 |
+
type=float,
|
| 184 |
+
default=0.0,
|
| 185 |
+
help="Penalty scale subtracted from composed residual candidates according to "
|
| 186 |
+
"their mean squared residual energy.",
|
| 187 |
+
)
|
| 188 |
parser.add_argument(
|
| 189 |
"--retrieval-residual-action-l2-penalty",
|
| 190 |
type=float,
|
|
|
|
| 314 |
retrieval_residual_source_advantage_bonus_scale=(
|
| 315 |
args.retrieval_residual_source_advantage_bonus_scale
|
| 316 |
),
|
| 317 |
+
retrieval_residual_composite_l2_penalty_scale=(
|
| 318 |
+
args.retrieval_residual_composite_l2_penalty_scale
|
| 319 |
+
),
|
| 320 |
retrieval_residual_action_l2_penalty=args.retrieval_residual_action_l2_penalty,
|
| 321 |
retrieval_residual_scale=args.retrieval_residual_scale,
|
| 322 |
retrieval_residual_scales=retrieval_residual_scales,
|
scripts/slurm/eval_maniskill_policy_rollout.sbatch
CHANGED
|
@@ -60,6 +60,7 @@ RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTAGE="${RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTA
|
|
| 60 |
RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
|
| 61 |
RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
|
| 62 |
RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
|
|
|
|
| 63 |
RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
|
| 64 |
RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
|
| 65 |
RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
|
|
@@ -138,6 +139,7 @@ apptainer exec --nv \
|
|
| 138 |
--retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
|
| 139 |
--retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
|
| 140 |
--retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
|
|
|
|
| 141 |
--retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
|
| 142 |
--retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
|
| 143 |
--retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
|
|
|
|
| 60 |
RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
|
| 61 |
RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
|
| 62 |
RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
|
| 63 |
+
RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE="${RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE:-0.0}"
|
| 64 |
RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
|
| 65 |
RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
|
| 66 |
RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
|
|
|
|
| 139 |
--retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
|
| 140 |
--retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
|
| 141 |
--retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
|
| 142 |
+
--retrieval-residual-composite-l2-penalty-scale "$RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE" \
|
| 143 |
--retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
|
| 144 |
--retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
|
| 145 |
--retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
|
scripts/slurm/eval_maniskill_policy_rollout_cpu_smoke.sbatch
CHANGED
|
@@ -59,6 +59,7 @@ RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTAGE="${RETRIEVAL_RESIDUAL_MIN_SOURCE_ADVANTA
|
|
| 59 |
RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
|
| 60 |
RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
|
| 61 |
RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
|
|
|
|
| 62 |
RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
|
| 63 |
RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
|
| 64 |
RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
|
|
@@ -134,6 +135,7 @@ apptainer exec \
|
|
| 134 |
--retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
|
| 135 |
--retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
|
| 136 |
--retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
|
|
|
|
| 137 |
--retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
|
| 138 |
--retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
|
| 139 |
--retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
|
|
|
|
| 59 |
RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE:-0.0}"
|
| 60 |
RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE:-0.0}"
|
| 61 |
RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE="${RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE:-0.0}"
|
| 62 |
+
RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE="${RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE:-0.0}"
|
| 63 |
RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY="${RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY:-0.0}"
|
| 64 |
RETRIEVAL_RESIDUAL_SCALE="${RETRIEVAL_RESIDUAL_SCALE:-1.0}"
|
| 65 |
RETRIEVAL_RESIDUAL_SCALES="${RETRIEVAL_RESIDUAL_SCALES:-}"
|
|
|
|
| 135 |
--retrieval-residual-source-progress-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_PROGRESS_BONUS_SCALE" \
|
| 136 |
--retrieval-residual-source-score-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_SCORE_BONUS_SCALE" \
|
| 137 |
--retrieval-residual-source-advantage-bonus-scale "$RETRIEVAL_RESIDUAL_SOURCE_ADVANTAGE_BONUS_SCALE" \
|
| 138 |
+
--retrieval-residual-composite-l2-penalty-scale "$RETRIEVAL_RESIDUAL_COMPOSITE_L2_PENALTY_SCALE" \
|
| 139 |
--retrieval-residual-action-l2-penalty "$RETRIEVAL_RESIDUAL_ACTION_L2_PENALTY" \
|
| 140 |
--retrieval-residual-scale "$RETRIEVAL_RESIDUAL_SCALE" \
|
| 141 |
--retrieval-residual-scales "$RETRIEVAL_RESIDUAL_SCALES" \
|
scripts/slurm/summarize_h16_policy_ckpt.sbatch
CHANGED
|
@@ -91,6 +91,9 @@ for result_path in sorted(base_dir.glob(f"seed_*/{out_name}")):
|
|
| 91 |
"retrieval_residual_source_advantage_bonus_scale": data.get(
|
| 92 |
"retrieval_residual_source_advantage_bonus_scale", 0.0
|
| 93 |
),
|
|
|
|
|
|
|
|
|
|
| 94 |
"retrieval_residual_scale": data.get("retrieval_residual_scale", 0.0),
|
| 95 |
"retrieval_residual_scales": data.get("retrieval_residual_scales", []),
|
| 96 |
"retrieval_residual_anchor": data.get("retrieval_residual_anchor", "none"),
|
|
|
|
| 91 |
"retrieval_residual_source_advantage_bonus_scale": data.get(
|
| 92 |
"retrieval_residual_source_advantage_bonus_scale", 0.0
|
| 93 |
),
|
| 94 |
+
"retrieval_residual_composite_l2_penalty_scale": data.get(
|
| 95 |
+
"retrieval_residual_composite_l2_penalty_scale", 0.0
|
| 96 |
+
),
|
| 97 |
"retrieval_residual_scale": data.get("retrieval_residual_scale", 0.0),
|
| 98 |
"retrieval_residual_scales": data.get("retrieval_residual_scales", []),
|
| 99 |
"retrieval_residual_anchor": data.get("retrieval_residual_anchor", "none"),
|
tests/test_maniskill_policy_rollout.py
CHANGED
|
@@ -1022,6 +1022,35 @@ def test_retrieval_residual_reducer_builds_composed_type_tangents() -> None:
|
|
| 1022 |
)
|
| 1023 |
|
| 1024 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1025 |
def test_retrieval_residual_reducer_penalizes_low_consensus_tangents() -> None:
|
| 1026 |
residuals, candidate_types, bonuses = _reduce_residual_candidates_by_type(
|
| 1027 |
[
|
|
|
|
| 1022 |
)
|
| 1023 |
|
| 1024 |
|
| 1025 |
+
def test_retrieval_residual_reducer_penalizes_composed_tangent_energy() -> None:
|
| 1026 |
+
residuals, candidate_types, bonuses = _reduce_residual_candidates_by_type(
|
| 1027 |
+
[
|
| 1028 |
+
[[0.0, 0.0]],
|
| 1029 |
+
[[0.4, 0.2]],
|
| 1030 |
+
[[-0.2, 0.0]],
|
| 1031 |
+
],
|
| 1032 |
+
[
|
| 1033 |
+
"policy_residual",
|
| 1034 |
+
"residual_no_op",
|
| 1035 |
+
"residual_wrong_gripper",
|
| 1036 |
+
],
|
| 1037 |
+
mode="compose_mean_by_type",
|
| 1038 |
+
bonuses=[0.0, 0.04, 0.02],
|
| 1039 |
+
composite_l2_penalty_scale=0.5,
|
| 1040 |
+
)
|
| 1041 |
+
|
| 1042 |
+
assert candidate_types == [
|
| 1043 |
+
"policy_residual",
|
| 1044 |
+
"residual_no_op",
|
| 1045 |
+
"residual_wrong_gripper",
|
| 1046 |
+
"residual_no_op+residual_wrong_gripper",
|
| 1047 |
+
]
|
| 1048 |
+
assert np.allclose(np.asarray(residuals[-1], dtype=np.float32), [[0.2, 0.2]])
|
| 1049 |
+
# Base composite bonus is mean(0.04, 0.02)=0.03; residual energy is
|
| 1050 |
+
# mean([0.2^2, 0.2^2])=0.04, so a 0.5 penalty subtracts 0.02.
|
| 1051 |
+
assert np.allclose(bonuses, [0.0, 0.04, 0.02, 0.01])
|
| 1052 |
+
|
| 1053 |
+
|
| 1054 |
def test_retrieval_residual_reducer_penalizes_low_consensus_tangents() -> None:
|
| 1055 |
residuals, candidate_types, bonuses = _reduce_residual_candidates_by_type(
|
| 1056 |
[
|