Add stack bowl RECAP value checkpoint metadata

#2
franka/rlinf/20260903-202954-franka_stack_bowl_pi05_recap_value_sft_fail300/README.md ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Franka Stack Bowl RECAP Value SFT Checkpoint
2
+
3
+ This checkpoint is a RECAP value/critic model trained for the Franka stack-bowl task using labeled rollout data.
4
+
5
+ Important: this is not a direct action-policy checkpoint. It is the RECAP value model checkpoint produced by `examples/offline_rl/advantage_labeling/recap/train_value.py`.
6
+
7
+ ## Run
8
+
9
+ - Model family in config: `pi05`
10
+ - Task: `Stack the three bowls in size order: the purple bowl first, then the beige bowl.`
11
+ - Dataset: stack-bowl RECAP rollout labels, 34 episodes, 196579 frames
12
+ - Labels: 22 success / 12 failure
13
+ - Return sidecar: `returns_fail300.parquet`
14
+ - Training: resumed from `global_step_5000`, completed to `global_step_8000`
15
+ - Final checkpoint: `global_step_8000`
16
+
17
+ ## Final eval metrics
18
+
19
+ - `eval/loss`: 2.1
20
+ - `eval/mae`: 0.0219
21
+ - `eval/cat_acc_best`: 0.287
22
+ - `eval/cat_acc_neighbor`: 0.44
23
+
24
+ ## Files
25
+
26
+ - `checkpoints/global_step_8000/actor/model_state_dict/full_weights.pt`: full actor/value model state dict
27
+ - `run_value_sft_fail300_resume5000.sh`: resume training command
28
+ - `value_sft_fail300_resume5000_mb2_tmux.log`: training log from the successful resume segment
franka/rlinf/20260903-202954-franka_stack_bowl_pi05_recap_value_sft_fail300/checkpoints/global_step_8000/actor/model_state_dict/full_weights.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3c42f0df0c54fa5003707307333cdb040c9a653335279a03aae487f1aa6653c6
3
+ size 1799192037
franka/rlinf/20260903-202954-franka_stack_bowl_pi05_recap_value_sft_fail300/logs/value_sft_fail300_resume5000_mb2_tmux.log ADDED
The diff for this file is too large to render. See raw diff
 
franka/rlinf/20260903-202954-franka_stack_bowl_pi05_recap_value_sft_fail300/run_manifest.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "checkpoint_type": "recap_value_sft",
3
+ "policy_executable": false,
4
+ "model_type": "pi05",
5
+ "robot_type": "franka",
6
+ "task": "stack_bowls",
7
+ "prompt": "Stack the three bowls in size order: the purple bowl first, then the beige bowl.",
8
+ "dataset_path_on_amax": "/data2/yangky/test/recap_stack_bowls/datasets/stack_bowls_recap_rollout_rc",
9
+ "num_episodes": 34,
10
+ "num_frames": 196579,
11
+ "success_episodes": 22,
12
+ "failure_episodes": 12,
13
+ "returns_tag": "fail300",
14
+ "return_min": -20713,
15
+ "return_max": 0,
16
+ "max_steps": 8000,
17
+ "resume_from": "global_step_5000",
18
+ "final_checkpoint": "global_step_8000",
19
+ "eval_loss": 2.1,
20
+ "eval_mae": 0.0219,
21
+ "eval_cat_acc_best": 0.287,
22
+ "eval_cat_acc_neighbor": 0.44
23
+ }
franka/rlinf/20260903-202954-franka_stack_bowl_pi05_recap_value_sft_fail300/run_value_sft_fail300_resume5000.sh ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ set -euo pipefail
2
+ source /data/lfwj/miniconda3/etc/profile.d/conda.sh
3
+ conda activate lerobot
4
+ cd /data/yangky/test/RealWorld-RLinf
5
+ export RUN_ROOT=/data2/yangky/test/recap_stack_bowls
6
+ export REPO_PATH=/data/yangky/test/RealWorld-RLinf
7
+ export OPENPI_ARCH=/home/amax/.cache/uv/archive-v0/ZQ6tV-bbWGFB5mqO
8
+ export PYTHONNOUSERSITE=1
9
+ export PYTHONPATH=$OPENPI_ARCH:$REPO_PATH:/data/linjianqi/RoboTwin/policy/pi0/packages/openpi-client/src:${PYTHONPATH:-}
10
+ export LD_LIBRARY_PATH=$CONDA_PREFIX/lib:${LD_LIBRARY_PATH:-}
11
+ export CUDA_VISIBLE_DEVICES=0
12
+ export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
13
+ export HF_HOME=$RUN_ROOT/hf_home
14
+ export HF_DATASETS_CACHE=$RUN_ROOT/hf_datasets_cache
15
+ export TRANSFORMERS_CACHE=$RUN_ROOT/hf_home/transformers
16
+ export MUJOCO_GL=egl
17
+ export PYOPENGL_PLATFORM=egl
18
+ export AV_LOG_FORCE_NOCOLOR=1
19
+ export LIBAV_LOG_LEVEL=quiet
20
+ export OPENCV_LOG_LEVEL=off
21
+ ray stop -f || true
22
+ mkdir -p "/data2/yangky/test/rsb_ray" "$HF_HOME" "$HF_DATASETS_CACHE" "$TRANSFORMERS_CACHE"
23
+ ray start --head --include-dashboard=true --dashboard-host=127.0.0.1 --num-gpus=1 --temp-dir="/data2/yangky/test/rsb_ray" --disable-usage-stats
24
+ python examples/offline_rl/advantage_labeling/recap/train_value.py --config-path /data/yangky/test/RealWorld-RLinf/examples/offline_rl/config --config-name recap_value_model_sft \
25
+ data.tag=fail300 +data.return_min=-20713 +data.return_max=0 \
26
+ data.train_data_paths='[{dataset_path: /data2/yangky/test/recap_stack_bowls/datasets/stack_bowls_recap_rollout_rc, type: rollout, weight: 1.0, robot_type: franka, model_type: pi05}]' \
27
+ data.eval_data_paths='[{dataset_path: /data2/yangky/test/recap_stack_bowls/datasets/stack_bowls_recap_rollout_rc, max_samples: 4096, robot_type: franka, model_type: pi05}]' \
28
+ data.robot_type=franka data.model_type=pi05 data.action_dim=7 data.train_num_workers=2 data.eval_num_workers=1 \
29
+ actor.micro_batch_size=2 actor.global_batch_size=16 actor.model.action_dim=7 \
30
+ actor.model.siglip_path=/data2/yangky/test/recap_stack_bowls/models/recap_value_backbones/siglip2-so400m-patch14-224 \
31
+ actor.model.gemma3_path=/data2/yangky/test/recap_stack_bowls/models/recap_value_backbones/gemma-3-270m \
32
+ actor.model.tokenizer_path=/data2/yangky/test/recap_stack_bowls/models/recap_value_backbones/gemma-3-270m \
33
+ runner.max_steps=8000 runner.save_interval=1000 runner.val_check_interval=500 actor.optim.total_training_steps=8000 \
34
+ \
35
+ +runner.resume_dir=/data2/yangky/test/recap_stack_bowls/logs/recap_stack_bowls_training/value_sft/stack_bowls_recap_value_sft_fail300/checkpoints/global_step_5000 \
36
+ runner.logger.log_path=/data2/yangky/test/recap_stack_bowls/logs/recap_stack_bowls_training/value_sft \
37
+ runner.logger.experiment_name=stack_bowls_recap_value_sft_fail300