twanghcmut's picture
download
raw
6.7 kB
{
"target_model": "nvidia/Cosmos-Transfer2.5-2B",
"prompt": "A Franka Panda robot arm with a black two-finger gripper works at a cluttered wooden laboratory workbench. The arm picks up a small blue plastic brick from the bench top, lifts it, carries it smoothly to the right, and sets it down on top of a tall stack of hardcover and paperback books sitting on the bench, then releases the brick and draws back. Behind the bench is a blue-painted pegboard wall with an aluminium extrusion frame and a yellow bar. On the bench are stacked books, a white plastic bin holding black binders, a white mug of pens, a cardboard box, and a black tool case. Below the bench top is a dark steel tool chest with one drawer pulled open. The room is lit by even overhead fluorescent light. The camera is fixed on a tripod and does not move. Photorealistic, sharp focus, natural indoor lighting, real laboratory footage.",
"prompt_words": 151,
"resolution": [
1280,
720
],
"frames": 186,
"fps": 16,
"duration_s": 11.625,
"frames_is_multiple_of_93": true,
"source_render_dir": "outputs/conditioning_cosmos_720p",
"source_render_meta": {
"no_dynamics": "Kinematic replay only: no mass, friction, or inertia is modelled. A grasp is a rigid attachment of the object to the gripper frame, not a contact/force simulation. Commanded speed changes only the timing of the motion, never its path.",
"single_viewpoint_capture": "Applies to the DEPTH channel only in this bundle. Background depth is still a single-viewpoint 2.5D capture, so surfaces the camera never saw have no points and any view far from the capture pose exposes real holes. Background *appearance* no longer has this limitation: rgb/vis/edge take their background from a real photograph (temporal median of the episode), not from the cloud.",
"no_collision_checking": "No collision checking was performed between the robot, the manipulated object(s), or the static scene, at any stage of producing this trajectory or this render.",
"no_settling_at_release": "SUPERSEDED for this bundle: the released object WAS settled under gravity (MuJoCo 3.11.0). Its tilt at release was 37.26 deg and 0.09 deg after settling, reached in 0.389 s of simulated time. Mass is not load-bearing: a rigid body's fall-and-settle trajectory is mass-independent, which was verified by re-running at 10x mass. Everything else in no_dynamics still holds -- the transport motion itself is kinematic."
},
"episode": "AUTOLab+0d4edc83+2023-10-21-19h-07m-04s",
"camera": "ext1",
"camera_serial": "22008760",
"camera_static_max_extrinsic_deviation": 0.0,
"background": {
"mode": "plate",
"method": "per-pixel temporal median of all real frames of the episode video, which removes the moving arm and leaves a robot-free photographic plate. Valid only because the capture camera is static (checked above)."
},
"depth": {
"convention": "relative INVERSE depth, near = bright, 8-bit -- matching what DepthAnything (Cosmos's own depth extractor) produces, NOT metric depth. The metric 16-bit millimetre PNGs in the render dir remain the source of truth.",
"normalization": "global",
"global_inverse_range_per_m": [
0.5924170613288879,
4.329004287719727
],
"global_range_m": [
0.2310000022029877,
1.6880000008049012
],
"why_global": "Per-frame normalization makes the control video flicker as the scene's near/far extremes change, and the model reproduces that as brightness pumping in the output.",
"hole_fill": "pyramid push-pull (smooth interpolation), measured pixels kept exact, applied before inversion. NOT nearest-neighbour: that is a Voronoi partition and tessellates the filled area into flat polygonal facets, which the model renders as fake glass-like planes.",
"hole_fraction_mean": 0.17996547612380526,
"hole_fraction_max": 0.20676106770833336,
"min_background_depth_m": 0.4,
"background_readings_rejected": 4769257,
"background_readings_rejected_fraction": 0.029300296912311383,
"why_reject_near_background": "The ZED's minimum reported depth in this capture is 233 mm. Stereo matching failure saturates disparity and therefore lands at the sensor's NEAR limit, so background readings pinned just above that minimum -- concentrated in the dark curtain and the far wall -- are failures, not measurements. In an inverse-depth control they are the brightest pixels in the frame: they draw a hard bogus structure across the upper frame and compress the range available to everything real. Foreground is exempt: the arm genuinely reaches within 30 cm of this wide-angle camera, and its depth is rendered from the mesh, not measured.",
"background_depth_native_resolution_note": "Background depth is unprojected from a single ~320x180 capture, so it is smooth/blocky relative to the 1280x720 output. Robot and object depth comes from the mesh renderer at full output resolution and is sharp."
},
"vis": {
"content": "the composited RGB (real plate + CG robot/object), Gaussian blurred",
"blur_sigma_px": 18.0,
"kernel_px": 109,
"not_used_by_default": "The generated specs give the vis branch a preset_blur_strength instead of this file, so the repo blurs video_path itself with the strength the branch was trained on. This file's sigma is a hand-picked guess at that distribution; it is exported only for A/B."
},
"segmentation": {
"id_map": {
"0": "background_scene",
"1": "robot",
"2": "brick"
},
"palette": {
"0": [
0,
0,
0
],
"1": [
0,
200,
255
],
"2": [
255,
80,
0
],
"3": [
80,
255,
80
],
"4": [
255,
0,
200
],
"5": [
255,
220,
0
]
}
},
"control_weights": {
"vis": 0.4,
"depth": 0.9,
"guidance": 3.0,
"rationale": "Depth is where our real information is -- foreground depth is exact mesh z-buffer. vis is the strongest appearance control, so it starts low; raising it turns the output into a de-blurred CG render.",
"mask_semantics": "cosmos-transfer2.5's mask_path is BINARY (white = apply this control here), not a per-pixel weight map. Per-control strength is the scalar control_weight only."
},
"video_encoding": {
"codec": "libx264",
"pix_fmt": "yuv420p",
"crf": 12,
"gop": 16,
"scene_cut_detection": false
},
"spec_schema_source": "third_party/cosmos-transfer2.5/assets/robot_example/*/*_spec.json -- the repo's own shipped examples, read directly rather than inferred from documentation."
}

Xet Storage Details

Size:
6.7 kB
·
Xet hash:
3ca0f72457a86d4bb1c8241acaf3bce7cc2fe795e461a97e0cc18a0b1421b57d

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.