marl-world-model / config.json
Devin
Add README.md and config.json with architecture/metadata
f4ec55f
Raw
History Blame Contribute Delete
2.93 kB
{
"model_type": "marl-ippo-actor-critic",
"name": "MARL IPPO Warehouse Cube Sorting Policy",
"description": "Independent PPO policies for 2-4 mobile warehouse robots in NVIDIA Isaac Sim, with optional world-model imagination.",
"task": "multi-agent-warehouse-cube-sorting",
"simulator": {
"name": "NVIDIA Isaac Sim",
"versions_tested": ["5.1.0", "6.0.1 pip"],
"headless": true,
"renderer": "D3D12"
},
"num_agents": 2,
"observation": {
"image": {
"shape": [3, 128, 128],
"dtype": "uint8",
"description": "RGB camera image per agent, stored/transposed to CHW for the policy network."
},
"state": {
"shape": [12],
"dtype": "float32",
"description": "[x, y, yaw, vx, vy, omega, nearest_cube_dx, nearest_cube_dy, nearest_cube_dist, target_bin_dx, target_bin_dy, target_bin_dist]"
}
},
"action_space": {
"type": "continuous",
"shape": [4],
"low": -1.0,
"high": 1.0,
"description": "[forward, lateral, angular, gripper]",
"gripper_threshold": 0.5
},
"policy": {
"algorithm": "IPPO",
"architecture": {
"image_encoder": {
"type": "CNN",
"input_shape": [3, 128, 128],
"conv_layers": [
{"in_channels": 3, "out_channels": 32, "kernel_size": 3, "stride": 2, "padding": 1},
{"in_channels": 32, "out_channels": 64, "kernel_size": 3, "stride": 2, "padding": 1},
{"in_channels": 64, "out_channels": 64, "kernel_size": 3, "stride": 2, "padding": 1}
],
"pooling": "AdaptiveAvgPool2d(4, 4)"
},
"shared_mlp": {
"input_dim": "conv_flat + state_dim",
"hidden_dim": 256,
"num_layers": 2,
"activation": "ReLU"
},
"value_head": {"input_dim": 256, "output_dim": 1},
"actor_head": {
"type": "gaussian",
"input_dim": 256,
"output_dim": 4
},
"state_dim": 12,
"hidden_dim": 256,
"cnn_channels": [32, 64, 64],
"reduce_spatial": 4
},
"hyperparameters": {
"lr": 3e-4,
"gamma": 0.99,
"gae_lambda": 0.95,
"clip_eps": 0.2,
"value_loss_coef": 0.5,
"entropy_coef": 0.01,
"update_epochs": 4,
"batch_size": 64,
"max_grad_norm": 0.5,
"normalize_advantages": true
}
},
"world_model": {
"backends": ["mock", "cosmos3", "dreamdojo"],
"default": "mock",
"fallback_to_mock": true,
"imagination_horizon_default": 20
},
"so101_bridge": {
"input_image_size": [128, 128],
"state_dim_policy": 12,
"action_mode": "delta",
"max_delta": 0.05,
"joint_keys": [
"shoulder_pan.pos",
"shoulder_lift.pos",
"elbow_flex.pos",
"wrist_flex.pos",
"wrist_roll.pos",
"gripper.pos"
]
},
"lerobot_dataset_repo": "Ryukijano/marl-world-model-lerobot",
"source_repo": "Ryukijano/marl-world-model",
"license": "apache-2.0"
}