# Data config: fruit_pick, 100 real-robot demos, Franka panda, 2-cam. # Downloaded 2026-08-09 from # https://huggingface.co/datasets/aabyaneh/fruit_pick -- LeRobot codebase_version # v3.0 (file-chunked, NOT the v2.1 per-episode-file layout our own # lift2lerobot converter produces), confirmed the vendored lerobot loader here # (fastwam/datasets/lerobot/lerobot/lerobot_dataset.py) supports v3.0 natively, # so no reconversion needed. # # SCHEMA VERIFIED AGAINST meta/info.json (2026-08-09), differs from every # other dataset in this repo in one way: # - camera keys are "agentview"/"wrist" (not "image"/"wrist_image") # # observation.state came off HF as a native 15-dim vector (eef_xyz[3] + # eef_quat_xyzw[4] + joint_0..6[7] + gripper_width[1]) -- NOT the 8-dim # eef_pose+gripper convention every other dataset here uses. Per user # request (this run must match place_cube_in_bowl/lift_new exactly), # fastwam_train/scripts/fix_fruit_pick_state_schema.py rewrote the LOCAL # parquet's observation.state down to the SAME 8-dim formula # lift2lerobot/convert_lift_hdf5_to_lerobot_v21.py uses: eef_pos(3) + # quat2axisangle(eef_quat)(3) + [gripper_width/2, -gripper_width/2] -- # bit-identical convention, not a from-scratch reimplementation. Original # 15-dim parquet backed up to data/chunk-000/file-000.parquet.orig_15dim_state. # # action is 7-dim (delta_xyz, delta_rot_xyz, grasp), SAME as every other # task's schema/mask already; delta_rot_x/y/z are exactly 0.0 across all # 20,149 frames (verified), matching the no-rotation convention used # everywhere else in this pipeline. No changes needed there. # # Task prompt: "Lift the lid, put it aside, and pick the black plum." -- the # dataset's own meta/tasks.parquet as downloaded said "pick up the fruit and # place it in the basket" instead; user chose to keep their own prompt, so the # LOCAL copy of data/fruit_pick/meta/tasks.parquet was overwritten to match # (original backed up to tasks.parquet.orig_hf_label) -- the HF repo itself # still has the old label. Must match the persistent T5 text cache exactly # (see fastwam_train/scripts/precompute_text_embeds.sh). # # val_set_proportion 0.0 -- NO validation (user decision 2026-08-09: use the # entire 100-episode dataset for training, no held-out split). No # `episode_indices` either -- the full 100 episodes are used as-is; there's # no larger pool to subset from (unlike George's pcbnew100, which slices 100 # out of place_cube_new's 500), so no capping is needed. # # LOCAL-DISK paths: dataset_dirs and text_embedding_cache_dir point at /dev/shm # copies staged by fastwam_train/scripts/stage_fruit_pick_shm.sh -- stage # BEFORE launching; re-stage after any node reboot. Persistent copy: # data/fruit_pick/ (LeRobot v3.0, 100 episodes, 20,149 frames). train: _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset dataset_dirs: - /dev/shm/fruit_pick_lerobot_v21 shape_meta: images: - key: agentview raw_shape: [3, 256, 256] shape: [3, 224, 224] - key: wrist raw_shape: [3, 256, 256] shape: [3, 224, 224] action: - key: default raw_shape: 7 # before transform shape: 7 # after transform state: - key: default raw_shape: 8 # before transform -- eef_pos(3)+axisangle(3)+gripper(2), see header shape: 8 # after transform num_frames: 33 global_sample_stride: 1 action_video_freq_ratio: 4 # 32 action, 9 video frames video_size: [224, 448] # final resize video camera_key: null val_set_proportion: 0.0 is_training_set: true skip_padding_as_possible: false concat_multi_camera: "horizontal" processor: _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor shape_meta: ${data.train.shape_meta} num_obs_steps: ${data.train.num_frames} num_output_cameras: 2 action_output_dim: 7 # eef delta xyz (3) + delta rot xyz (3, always 0) + grasp (1) proprio_output_dim: 8 # eef_pos(3) + axisangle(3) + gripper(2) -- matches place_cube/lift_new delta_action_dim_mask: default: [true, true, true, true, true, true, false] # eef poses are delta, gripper is not # action & state normalization action_state_transforms: null use_stepwise_action_norm: False norm_default_mode: min/max norm_exception_mode: null action_state_merger: _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign train_transforms: - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: [224, 224] val_transforms: - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: [224, 224] text_embedding_cache_dir: /dev/shm/fruit_pick_text_embeds_cache context_len: 128