| """ |
| MuJoCo Data Generator |
| ===================== |
| Produces exact dataset files for all 4 training phases. |
| Run: python generate_data.py --phase all --n_scenes 500 --output_dir data/ |
| |
| Dependencies: mujoco, numpy, json (stdlib) |
| NO torch dependency — this is pure data generation. |
| |
| Output structure: |
| data/ |
| phase1/ # Encoder training (supervised) |
| scene_0000.npz # Per-scene: image, gt_poses, gt_masks, gt_contacts, gt_sdf, gt_materials |
| scene_0001.npz |
| ... |
| manifest.json # List of all scenes with metadata |
| |
| phase2/ # Vectorizer training (contrastive) |
| pair_0000.npz # Per-pair: phi_g_A, phi_g_B (same scene, 2 cameras) |
| ... |
| manifest.json |
| |
| phase3a/ # Cross-encoder alignment (contrastive) |
| sample_0000.json # Per-sample: phi_g, text_description, tokenized_text |
| ... |
| manifest.json |
| |
| phase3b/ # Action prediction (behavioral cloning) |
| demo_0000.npz # Per-step: phi_g, image, text_instruction, gt_action |
| ... |
| manifest.json |
| """ |
|
|
| import mujoco |
| import numpy as np |
| import json |
| import os |
| import argparse |
| from pathlib import Path |
|
|
|
|
| |
| |
| |
|
|
| MATERIALS = [ |
| {"name": "wood", "mass": (0.3, 0.8), "friction": 0.4, "density": 600, "color": (0.65, 0.45, 0.25)}, |
| {"name": "rubber", "mass": (0.2, 0.6), "friction": 0.8, "density": 1100, "color": (0.18, 0.18, 0.22)}, |
| {"name": "metal", "mass": (1.0, 5.0), "friction": 0.2, "density": 7800, "color": (0.72, 0.73, 0.76)}, |
| {"name": "plastic", "mass": (0.05, 0.3), "friction": 0.35,"density": 1200, "color": (0.20, 0.55, 0.85)}, |
| {"name": "glass", "mass": (0.2, 0.5), "friction": 0.15,"density": 2500, "color": (0.80, 0.85, 0.90)}, |
| {"name": "foam", "mass": (0.01,0.05), "friction": 0.3, "density": 30, "color": (0.90, 0.85, 0.40)}, |
| ] |
|
|
| SHAPES = [ |
| {"type": "box", "size_template": "0.{s1} 0.{s2} 0.{s3}", "sdf_fn": "box"}, |
| {"type": "sphere", "size_template": "0.{r}", "sdf_fn": "sphere"}, |
| {"type": "cylinder", "size_template": "0.{r} 0.{h}", "sdf_fn": "cylinder"}, |
| ] |
|
|
| CAMERA_POSITIONS = [ |
| {"name": "front", "pos": "0 -0.5 0.7", "xyaxes": "1 0 0 0 0.5 0.87"}, |
| {"name": "front_far", "pos": "0 -0.7 0.8", "xyaxes": "1 0 0 0 0.5 0.87"}, |
| {"name": "left", "pos": "-0.5 -0.2 0.65","xyaxes": "0.4 1 0 -0.5 0.2 0.85"}, |
| {"name": "right", "pos": "0.5 -0.2 0.65", "xyaxes": "-0.4 1 0 0.5 0.2 0.85"}, |
| {"name": "top", "pos": "0 0 1.2", "xyaxes": "1 0 0 0 1 0"}, |
| {"name": "angle1", "pos": "0.3 -0.4 0.65", "xyaxes": "0.8 0.6 0 -0.3 0.4 0.87"}, |
| {"name": "angle2", "pos": "-0.3 -0.4 0.65","xyaxes": "0.8 -0.6 0 0.3 0.4 0.87"}, |
| ] |
|
|
|
|
| def generate_scene_xml(n_objects, camera_names=None, seed=None): |
| """Generate random MuJoCo scene XML + ground truth metadata. |
| |
| Returns: (xml_string, gt_metadata_dict) |
| """ |
| if seed is not None: |
| np.random.seed(seed) |
| |
| if camera_names is None: |
| camera_names = ["front", "top"] |
| |
| n_objects = np.random.randint(2, n_objects + 1) if isinstance(n_objects, int) and n_objects > 2 else n_objects |
| |
| objects_xml = "" |
| gt_objects = [] |
| |
| for i in range(n_objects): |
| mat = MATERIALS[np.random.randint(len(MATERIALS))] |
| shape_info = SHAPES[np.random.randint(len(SHAPES))] |
| mass = np.random.uniform(*mat["mass"]) |
| |
| |
| x = np.random.uniform(-0.18, 0.18) |
| y = np.random.uniform(-0.12, 0.12) |
| z = 0.45 |
| |
| |
| if shape_info["type"] == "box": |
| s = np.random.uniform(0.02, 0.04, 3) |
| size_str = f"{s[0]:.3f} {s[1]:.3f} {s[2]:.3f}" |
| half_extents = s.tolist() |
| elif shape_info["type"] == "sphere": |
| r = np.random.uniform(0.015, 0.035) |
| size_str = f"{r:.3f}" |
| half_extents = [r] |
| else: |
| r = np.random.uniform(0.015, 0.03) |
| h = np.random.uniform(0.02, 0.04) |
| size_str = f"{r:.3f} {h:.3f}" |
| half_extents = [r, h] |
| |
| c = mat["color"] |
| |
| cv = np.clip(np.array(c) + np.random.uniform(-0.1, 0.1, 3), 0, 1) |
| |
| objects_xml += f''' |
| <body name="obj_{i}" pos="{x:.4f} {y:.4f} {z:.4f}"> |
| <joint type="free"/> |
| <geom name="geom_{i}" type="{shape_info['type']}" size="{size_str}" |
| mass="{mass:.4f}" rgba="{cv[0]:.3f} {cv[1]:.3f} {cv[2]:.3f} 1" |
| friction="{mat['friction']} 0.005 0.0001"/> |
| </body>''' |
| |
| gt_objects.append({ |
| "index": i, |
| "name": f"obj_{i}", |
| "geom_name": f"geom_{i}", |
| "material": mat["name"], |
| "shape": shape_info["type"], |
| "sdf_type": shape_info["sdf_fn"], |
| "half_extents": half_extents, |
| "mass": float(mass), |
| "friction": float(mat["friction"]), |
| "density": float(mat["density"]), |
| "initial_pos": [float(x), float(y), float(z)], |
| "color": cv.tolist(), |
| }) |
| |
| cameras_xml = "" |
| for cn in camera_names: |
| cam = next(c for c in CAMERA_POSITIONS if c["name"] == cn) |
| cameras_xml += f'\n <camera name="{cn}" pos="{cam["pos"]}" xyaxes="{cam["xyaxes"]}" fovy="50"/>' |
| |
| xml = f"""<mujoco model="training_scene"> |
| <option timestep="0.002" gravity="0 0 -9.81"/> |
| <visual> |
| <global offwidth="256" offheight="256"/> |
| <headlight ambient="0.3 0.3 0.3" diffuse="0.6 0.6 0.6"/> |
| </visual> |
| <worldbody> |
| <light pos="0.3 -0.5 1.5" dir="-0.1 0.3 -1" castshadow="true"/> |
| <light pos="-0.3 0.3 1.0" dir="0.2 -0.2 -1" diffuse="0.3 0.3 0.35"/> |
| <geom name="floor" type="plane" size="1 1 0.1" rgba="0.4 0.4 0.43 1"/> |
| <body name="table" pos="0 0 0.38"> |
| <geom name="table_top" type="box" size="0.35 0.25 0.02" rgba="0.45 0.35 0.25 1"/> |
| </body> |
| {objects_xml} |
| {cameras_xml} |
| </worldbody> |
| </mujoco>""" |
| |
| return xml, gt_objects |
|
|
|
|
| |
| |
| |
|
|
| def sdf_box(x, half_extents): |
| """Signed distance to axis-aligned box at origin.""" |
| he = np.array(half_extents) |
| q = np.abs(x) - he |
| return np.linalg.norm(np.maximum(q, 0), axis=-1) + np.minimum(np.max(q, axis=-1), 0) |
|
|
| def sdf_sphere(x, radius): |
| return np.linalg.norm(x, axis=-1) - radius |
|
|
| def sdf_cylinder(x, radius, half_height): |
| dr = np.sqrt(x[..., 0]**2 + x[..., 1]**2) - radius |
| dh = np.abs(x[..., 2]) - half_height |
| return np.sqrt(np.maximum(dr, 0)**2 + np.maximum(dh, 0)**2) + np.minimum(np.maximum(dr, dh), 0) |
|
|
| def compute_sdf(x_local, sdf_type, half_extents): |
| if sdf_type == "box": |
| return sdf_box(x_local, half_extents) |
| elif sdf_type == "sphere": |
| return sdf_sphere(x_local, half_extents[0]) |
| elif sdf_type == "cylinder": |
| return sdf_cylinder(x_local, half_extents[0], half_extents[1]) |
| return np.ones(len(x_local)) |
|
|
|
|
| |
| |
| |
|
|
| def extract_ground_truth(model, data, gt_objects, n_sdf_points=500): |
| """Extract all ground truth from a settled MuJoCo scene. |
| |
| Returns dict with exact arrays needed for training. |
| """ |
| n_obj = len(gt_objects) |
| |
| |
| gt_poses = np.zeros((n_obj, 7), dtype=np.float32) |
| for i, obj in enumerate(gt_objects): |
| bid = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_BODY, obj["name"]) |
| pos = data.xpos[bid].copy() |
| |
| mat = data.xmat[bid].reshape(3, 3) |
| |
| quat = np.zeros(4) |
| mujoco.mju_mat2Quat(quat, mat.flatten()) |
| gt_poses[i, :3] = pos |
| gt_poses[i, 3:] = quat |
| |
| |
| gt_existence = np.ones(n_obj, dtype=np.float32) |
| |
| |
| gt_contacts = np.zeros((n_obj, n_obj), dtype=np.float32) |
| for c_idx in range(data.ncon): |
| con = data.contact[c_idx] |
| g1, g2 = con.geom1, con.geom2 |
| |
| for i, obj_i in enumerate(gt_objects): |
| gid_i = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_GEOM, obj_i["geom_name"]) |
| for j, obj_j in enumerate(gt_objects): |
| if i == j: |
| continue |
| gid_j = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_GEOM, obj_j["geom_name"]) |
| if (g1 == gid_i and g2 == gid_j) or (g1 == gid_j and g2 == gid_i): |
| gt_contacts[i, j] = 1.0 |
| gt_contacts[j, i] = 1.0 |
| |
| table_gid = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_GEOM, "table_top") |
| for i, obj_i in enumerate(gt_objects): |
| gid_i = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_GEOM, obj_i["geom_name"]) |
| if (g1 == gid_i and g2 == table_gid) or (g1 == table_gid and g2 == gid_i): |
| pass |
| |
| |
| gt_materials = np.zeros((n_obj, 4), dtype=np.float32) |
| for i, obj in enumerate(gt_objects): |
| gt_materials[i] = [obj["mass"], obj["friction"], obj["density"], 0.3] |
| |
| |
| sdf_samples = [] |
| for i, obj in enumerate(gt_objects): |
| bid = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_BODY, obj["name"]) |
| obj_pos = data.xpos[bid].copy() |
| obj_mat = data.xmat[bid].reshape(3, 3) |
| |
| |
| extent = max(obj["half_extents"]) * 3 |
| pts_world = obj_pos + np.random.uniform(-extent, extent, (n_sdf_points, 3)).astype(np.float32) |
| |
| |
| pts_local = (pts_world - obj_pos) @ obj_mat |
| |
| |
| sdf_vals = compute_sdf(pts_local, obj["sdf_type"], obj["half_extents"]) |
| |
| for p, s in zip(pts_world, sdf_vals): |
| sdf_samples.append([p[0], p[1], p[2], float(s), i]) |
| |
| sdf_samples = np.array(sdf_samples, dtype=np.float32) |
| |
| return { |
| "gt_poses": gt_poses, |
| "gt_existence": gt_existence, |
| "gt_contacts": gt_contacts, |
| "gt_materials": gt_materials, |
| "sdf_samples": sdf_samples, |
| } |
|
|
|
|
| |
| |
| |
|
|
| def generate_text_description(gt_objects, style="detailed"): |
| """Auto-generate text description from GT scene properties. |
| |
| Styles: |
| "simple": "3 objects on table" |
| "detailed": "wood box (0.5kg, μ=0.4), rubber sphere (0.3kg, μ=0.8), ..." |
| "natural": "A wooden block sits next to a rubber ball on a table" |
| "task": "Pick up the heaviest object" |
| """ |
| n = len(gt_objects) |
| |
| if style == "simple": |
| shapes = [o["shape"] for o in gt_objects] |
| return f"{n} objects on table: " + ", ".join(shapes) |
| |
| elif style == "detailed": |
| parts = [] |
| for o in gt_objects: |
| parts.append(f"{o['material']} {o['shape']} ({o['mass']:.2f}kg, μ={o['friction']})") |
| return f"{n} objects: " + ", ".join(parts) |
| |
| elif style == "natural": |
| descs = [] |
| for o in gt_objects: |
| adj = {"wood": "wooden", "rubber": "rubber", "metal": "metal", |
| "plastic": "plastic", "glass": "glass", "foam": "foam"} |
| shape_noun = {"box": "block", "sphere": "ball", "cylinder": "cylinder"} |
| descs.append(f"a {adj.get(o['material'], o['material'])} {shape_noun.get(o['shape'], o['shape'])}") |
| if len(descs) == 1: |
| return f"{descs[0]} on a table" |
| return ", ".join(descs[:-1]) + f", and {descs[-1]} on a table" |
| |
| elif style == "task": |
| tasks = [ |
| f"pick up the {gt_objects[0]['material']} {gt_objects[0]['shape']}", |
| f"push the heaviest object to the right", |
| f"grasp the {gt_objects[-1]['shape']} gently", |
| f"move the {gt_objects[0]['material']} object away from the {gt_objects[-1]['material']} one", |
| ] |
| return tasks[np.random.randint(len(tasks))] |
| |
| return f"{n} objects on table" |
|
|
|
|
| def tokenize_simple(text, vocab_size=10000, max_len=64): |
| """Dead-simple word-level tokenizer. Replace with real tokenizer in production. |
| |
| Returns: (token_ids [max_len], attention_mask [max_len]) |
| """ |
| words = text.lower().replace(",", " ,").replace("(", " ( ").replace(")", " ) ").split() |
| tokens = [hash(w) % (vocab_size - 2) + 2 for w in words] |
| tokens = tokens[:max_len] |
| mask = [False] * len(tokens) + [True] * (max_len - len(tokens)) |
| tokens = tokens + [0] * (max_len - len(tokens)) |
| return np.array(tokens, dtype=np.int64), np.array(mask, dtype=bool) |
|
|
|
|
| |
| |
| |
|
|
| def scripted_reach(model, data, target_obj_idx, gt_objects): |
| """Generate a reach action toward target object. |
| Returns: action [6] = (dx, dy, dz, d_roll, d_pitch, grip) |
| """ |
| bid = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_BODY, gt_objects[target_obj_idx]["name"]) |
| target_pos = data.xpos[bid].copy() |
| |
| |
| gripper_pos = np.array([0.0, 0.0, 0.55]) |
| |
| |
| delta = target_pos - gripper_pos |
| delta_norm = delta / (np.linalg.norm(delta) + 1e-8) |
| step_size = 0.02 |
| |
| action = np.zeros(6, dtype=np.float32) |
| action[:3] = delta_norm * step_size |
| action[3:5] = 0.0 |
| action[5] = 0.0 |
| |
| return action |
|
|
|
|
| def scripted_grasp(model, data, target_obj_idx, gt_objects): |
| """Generate a grasp action. Assumes gripper is above object. |
| Returns: action [6] |
| """ |
| obj = gt_objects[target_obj_idx] |
| action = np.zeros(6, dtype=np.float32) |
| action[2] = -0.01 |
| |
| action[5] = min(1.0, obj["mass"] * 9.81 / (obj["friction"] * 2 + 0.01) / 20) |
| return action |
|
|
|
|
| def scripted_push(model, data, target_obj_idx, gt_objects, push_dir=None): |
| """Generate a push action. |
| Returns: action [6] |
| """ |
| if push_dir is None: |
| push_dir = np.random.randn(2) |
| push_dir = push_dir / (np.linalg.norm(push_dir) + 1e-8) |
| |
| action = np.zeros(6, dtype=np.float32) |
| action[0] = push_dir[0] * 0.02 |
| action[1] = push_dir[1] * 0.02 |
| action[5] = 0.3 |
| return action |
|
|
|
|
| |
| |
| |
|
|
| def generate_phase1_data(output_dir, n_scenes=100, n_sdf_points=500): |
| """Generate supervised encoder training data. |
| |
| Each scene → one .npz file containing: |
| image: [256, 256, 3] uint8 |
| gt_poses: [n_obj, 7] float32 |
| gt_existence: [n_obj] float32 |
| gt_contacts: [n_obj, n_obj] float32 |
| gt_materials: [n_obj, 4] float32 |
| sdf_samples: [M, 5] float32 (x, y, z, sdf_value, object_id) |
| n_objects: int |
| """ |
| os.makedirs(output_dir, exist_ok=True) |
| manifest = [] |
| |
| for scene_idx in range(n_scenes): |
| n_obj = np.random.randint(2, 6) |
| xml, gt_objects = generate_scene_xml(n_obj, camera_names=["front"], seed=scene_idx) |
| |
| try: |
| model = mujoco.MjModel.from_xml_string(xml) |
| data = mujoco.MjData(model) |
| mujoco.mj_forward(model, data) |
| for _ in range(300): |
| mujoco.mj_step(model, data) |
| except Exception as e: |
| print(f" Scene {scene_idx} failed: {e}") |
| continue |
| |
| gt = extract_ground_truth(model, data, gt_objects, n_sdf_points) |
| |
| |
| |
| image_placeholder = np.zeros((256, 256, 3), dtype=np.uint8) |
| |
| fname = f"scene_{scene_idx:04d}.npz" |
| np.savez_compressed( |
| os.path.join(output_dir, fname), |
| image=image_placeholder, |
| **gt, |
| n_objects=np.array(len(gt_objects)), |
| ) |
| |
| manifest.append({ |
| "file": fname, |
| "n_objects": len(gt_objects), |
| "objects": gt_objects, |
| "scene_seed": scene_idx, |
| }) |
| |
| if scene_idx % 20 == 0: |
| print(f" Phase 1: {scene_idx}/{n_scenes} scenes generated") |
| |
| with open(os.path.join(output_dir, "manifest.json"), "w") as f: |
| json.dump(manifest, f, indent=2) |
| |
| print(f" Phase 1 complete: {len(manifest)} scenes → {output_dir}") |
| return manifest |
|
|
|
|
| def generate_phase2_data(output_dir, n_pairs=100, n_sdf_points=300): |
| """Generate contrastive scene pairs for vectorizer training. |
| |
| Each pair → one .npz file containing: |
| gt_poses_A, gt_poses_B: [n_obj, 7] (same, from same scene) |
| gt_existence: [n_obj] |
| gt_contacts: [n_obj, n_obj] |
| gt_materials: [n_obj, 4] |
| sdf_samples: [M, 5] |
| camera_A, camera_B: str names |
| |
| Positive pair: same scene, different camera → same Φ+G → same s |
| """ |
| os.makedirs(output_dir, exist_ok=True) |
| manifest = [] |
| |
| for pair_idx in range(n_pairs): |
| n_obj = np.random.randint(2, 5) |
| |
| cam_idxs = np.random.choice(len(CAMERA_POSITIONS), 2, replace=False) |
| cam_A = CAMERA_POSITIONS[cam_idxs[0]]["name"] |
| cam_B = CAMERA_POSITIONS[cam_idxs[1]]["name"] |
| |
| xml, gt_objects = generate_scene_xml(n_obj, camera_names=[cam_A, cam_B], seed=pair_idx + 10000) |
| |
| try: |
| model = mujoco.MjModel.from_xml_string(xml) |
| data = mujoco.MjData(model) |
| mujoco.mj_forward(model, data) |
| for _ in range(300): |
| mujoco.mj_step(model, data) |
| except: |
| continue |
| |
| gt = extract_ground_truth(model, data, gt_objects, n_sdf_points) |
| |
| fname = f"pair_{pair_idx:04d}.npz" |
| np.savez_compressed( |
| os.path.join(output_dir, fname), |
| |
| gt_poses=gt["gt_poses"], |
| gt_existence=gt["gt_existence"], |
| gt_contacts=gt["gt_contacts"], |
| gt_materials=gt["gt_materials"], |
| sdf_samples=gt["sdf_samples"], |
| n_objects=np.array(len(gt_objects)), |
| camera_A=cam_A, |
| camera_B=cam_B, |
| ) |
| |
| |
| gt_objects_neg = [] |
| for obj in gt_objects: |
| obj_neg = obj.copy() |
| |
| new_mat = MATERIALS[np.random.randint(len(MATERIALS))] |
| while new_mat["name"] == obj["material"]: |
| new_mat = MATERIALS[np.random.randint(len(MATERIALS))] |
| obj_neg["material"] = new_mat["name"] |
| obj_neg["friction"] = new_mat["friction"] |
| obj_neg["density"] = new_mat["density"] |
| obj_neg["mass"] = np.random.uniform(*new_mat["mass"]) |
| gt_objects_neg.append(obj_neg) |
| |
| gt_neg_materials = np.array([[o["mass"], o["friction"], o["density"], 0.3] |
| for o in gt_objects_neg], dtype=np.float32) |
| |
| fname_neg = f"pair_{pair_idx:04d}_neg.npz" |
| np.savez_compressed( |
| os.path.join(output_dir, fname_neg), |
| gt_poses=gt["gt_poses"], |
| gt_existence=gt["gt_existence"], |
| gt_contacts=gt["gt_contacts"], |
| gt_materials=gt_neg_materials, |
| sdf_samples=gt["sdf_samples"], |
| n_objects=np.array(len(gt_objects)), |
| ) |
| |
| manifest.append({ |
| "positive_file": fname, |
| "negative_file": fname_neg, |
| "n_objects": len(gt_objects), |
| "camera_A": cam_A, |
| "camera_B": cam_B, |
| }) |
| |
| if pair_idx % 20 == 0: |
| print(f" Phase 2: {pair_idx}/{n_pairs} pairs generated") |
| |
| with open(os.path.join(output_dir, "manifest.json"), "w") as f: |
| json.dump(manifest, f, indent=2) |
| |
| print(f" Phase 2 complete: {len(manifest)} pairs → {output_dir}") |
| return manifest |
|
|
|
|
| def generate_phase3a_data(output_dir, n_samples=200): |
| """Generate (scene, text) pairs for contrastive alignment. |
| |
| Each sample → one .json file containing: |
| gt_poses, gt_existence, gt_contacts, gt_materials (serialized as lists) |
| text_descriptions: dict with 4 styles |
| tokenized: dict with token_ids and attention_mask per style |
| """ |
| os.makedirs(output_dir, exist_ok=True) |
| manifest = [] |
| |
| for sample_idx in range(n_samples): |
| n_obj = np.random.randint(2, 5) |
| xml, gt_objects = generate_scene_xml(n_obj, seed=sample_idx + 20000) |
| |
| try: |
| model = mujoco.MjModel.from_xml_string(xml) |
| data = mujoco.MjData(model) |
| mujoco.mj_forward(model, data) |
| for _ in range(300): |
| mujoco.mj_step(model, data) |
| except: |
| continue |
| |
| gt = extract_ground_truth(model, data, gt_objects, n_sdf_points=100) |
| |
| |
| texts = {} |
| tokenized = {} |
| for style in ["simple", "detailed", "natural", "task"]: |
| text = generate_text_description(gt_objects, style=style) |
| toks, mask = tokenize_simple(text) |
| texts[style] = text |
| tokenized[style] = {"token_ids": toks.tolist(), "attention_mask": mask.tolist()} |
| |
| sample = { |
| "gt_poses": gt["gt_poses"].tolist(), |
| "gt_existence": gt["gt_existence"].tolist(), |
| "gt_contacts": gt["gt_contacts"].tolist(), |
| "gt_materials": gt["gt_materials"].tolist(), |
| "n_objects": len(gt_objects), |
| "objects": gt_objects, |
| "text_descriptions": texts, |
| "tokenized": tokenized, |
| } |
| |
| fname = f"sample_{sample_idx:04d}.json" |
| with open(os.path.join(output_dir, fname), "w") as f: |
| json.dump(sample, f) |
| |
| manifest.append({"file": fname, "n_objects": len(gt_objects), "texts": texts}) |
| |
| if sample_idx % 50 == 0: |
| print(f" Phase 3a: {sample_idx}/{n_samples} samples generated") |
| |
| with open(os.path.join(output_dir, "manifest.json"), "w") as f: |
| json.dump(manifest, f, indent=2) |
| |
| print(f" Phase 3a complete: {len(manifest)} samples → {output_dir}") |
| return manifest |
|
|
|
|
| def generate_phase3b_data(output_dir, n_demos=100, steps_per_demo=10): |
| """Generate behavioral cloning demonstrations. |
| |
| Each demo → one .npz file containing: |
| gt_poses: [T, n_obj, 7] |
| gt_existence: [n_obj] |
| gt_contacts: [T, n_obj, n_obj] |
| gt_materials: [n_obj, 4] |
| actions: [T, 6] |
| text_instruction: str |
| tokenized_instruction: (token_ids, attention_mask) |
| task_type: str ("reach", "grasp", "push") |
| """ |
| os.makedirs(output_dir, exist_ok=True) |
| manifest = [] |
| |
| task_types = ["reach", "grasp", "push"] |
| |
| for demo_idx in range(n_demos): |
| n_obj = np.random.randint(2, 4) |
| xml, gt_objects = generate_scene_xml(n_obj, seed=demo_idx + 30000) |
| |
| try: |
| model = mujoco.MjModel.from_xml_string(xml) |
| data = mujoco.MjData(model) |
| mujoco.mj_forward(model, data) |
| for _ in range(300): |
| mujoco.mj_step(model, data) |
| except: |
| continue |
| |
| target_idx = np.random.randint(len(gt_objects)) |
| task = task_types[np.random.randint(len(task_types))] |
| |
| text = f"{task} the {gt_objects[target_idx]['material']} {gt_objects[target_idx]['shape']}" |
| toks, mask = tokenize_simple(text) |
| |
| all_poses = [] |
| all_contacts = [] |
| all_actions = [] |
| |
| gt_init = extract_ground_truth(model, data, gt_objects, n_sdf_points=100) |
| |
| for step in range(steps_per_demo): |
| |
| poses_t = np.zeros((len(gt_objects), 7), dtype=np.float32) |
| for i, obj in enumerate(gt_objects): |
| bid = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_BODY, obj["name"]) |
| poses_t[i, :3] = data.xpos[bid] |
| quat = np.zeros(4) |
| mujoco.mju_mat2Quat(quat, data.xmat[bid].reshape(3, 3).flatten()) |
| poses_t[i, 3:] = quat |
| |
| all_poses.append(poses_t) |
| all_contacts.append(gt_init["gt_contacts"].copy()) |
| |
| |
| if task == "reach": |
| action = scripted_reach(model, data, target_idx, gt_objects) |
| elif task == "grasp": |
| action = scripted_grasp(model, data, target_idx, gt_objects) |
| else: |
| action = scripted_push(model, data, target_idx, gt_objects) |
| |
| all_actions.append(action) |
| |
| |
| target_bid = mujoco.mj_name2id(model, mujoco.mjtObj.mjOBJ_BODY, gt_objects[target_idx]["name"]) |
| data.xfrc_applied[target_bid, :3] = action[:3] * 10 |
| mujoco.mj_step(model, data) |
| data.xfrc_applied[target_bid, :3] = 0 |
| |
| fname = f"demo_{demo_idx:04d}.npz" |
| np.savez_compressed( |
| os.path.join(output_dir, fname), |
| gt_poses=np.array(all_poses), |
| gt_existence=gt_init["gt_existence"], |
| gt_contacts=np.array(all_contacts), |
| gt_materials=gt_init["gt_materials"], |
| actions=np.array(all_actions), |
| sdf_samples=gt_init["sdf_samples"], |
| token_ids=toks, |
| attention_mask=mask, |
| task_type=task, |
| n_objects=np.array(len(gt_objects)), |
| ) |
| |
| manifest.append({ |
| "file": fname, |
| "task": task, |
| "target": gt_objects[target_idx]["name"], |
| "text": text, |
| "n_objects": len(gt_objects), |
| "n_steps": steps_per_demo, |
| }) |
| |
| if demo_idx % 20 == 0: |
| print(f" Phase 3b: {demo_idx}/{n_demos} demos generated") |
| |
| with open(os.path.join(output_dir, "manifest.json"), "w") as f: |
| json.dump(manifest, f, indent=2) |
| |
| print(f" Phase 3b complete: {len(manifest)} demos → {output_dir}") |
| return manifest |
|
|
|
|
| |
| |
| |
|
|
| if __name__ == "__main__": |
| parser = argparse.ArgumentParser(description="Generate training data from MuJoCo") |
| parser.add_argument("--phase", default="all", choices=["1", "2", "3a", "3b", "all"]) |
| parser.add_argument("--output_dir", default="data") |
| parser.add_argument("--n_scenes", type=int, default=100) |
| args = parser.parse_args() |
| |
| print("=" * 60) |
| print("MuJoCo Data Generator for Φ+G Pipeline") |
| print("=" * 60) |
| |
| if args.phase in ["1", "all"]: |
| print(f"\n--- Phase 1: Encoder Training Data ({args.n_scenes} scenes) ---") |
| generate_phase1_data(f"{args.output_dir}/phase1", args.n_scenes) |
| |
| if args.phase in ["2", "all"]: |
| print(f"\n--- Phase 2: Vectorizer Training Data ({args.n_scenes} pairs) ---") |
| generate_phase2_data(f"{args.output_dir}/phase2", args.n_scenes) |
| |
| if args.phase in ["3a", "all"]: |
| print(f"\n--- Phase 3a: Alignment Data ({args.n_scenes * 2} samples) ---") |
| generate_phase3a_data(f"{args.output_dir}/phase3a", args.n_scenes * 2) |
| |
| if args.phase in ["3b", "all"]: |
| print(f"\n--- Phase 3b: Demonstration Data ({args.n_scenes} demos) ---") |
| generate_phase3b_data(f"{args.output_dir}/phase3b", args.n_scenes) |
| |
| print("\n" + "=" * 60) |
| print("Data generation complete!") |
| print(f"Output: {args.output_dir}/") |
| print("=" * 60) |
|
|