{ "_class_name": "CogVideoXTransformer3DMOTModel", "_diffusers_version": "0.34.0.dev0", "_name_or_path": "/mnt/bn/cx-space-hl03/yuxuan/reference_video_gen_outputs/train_multi_node_v5_alignment>10/model_weights/010000/transformer", "ablation_residual_addition": false, "ablation_single_encoder": false, "activation_fn": "gelu-approximate", "attention_bias": true, "attention_head_dim": 64, "attention_head_dim_mot_ref": null, "block_idx_with_mot_ref": [ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40 ], "dropout": 0.0, "flip_sin_to_cos": true, "freq_shift": 0, "in_channels": 32, "max_text_seq_length": 226, "norm_elementwise_affine": true, "norm_eps": 1e-05, "num_attention_heads": 48, "num_layers": 42, "num_ref_embeddings": null, "ofs_embed_dim": null, "out_channels": 16, "patch_bias": true, "patch_size": 2, "patch_size_t": null, "reference_train_mode": null, "sample_frames": 49, "sample_height": 60, "sample_width": 90, "spatial_interpolation_scale": 1.875, "supported_effect_types": null, "temporal_compression_ratio": 4, "temporal_interpolation_scale": 1.0, "text_embed_dim": 4096, "time_embed_dim": 512, "timestep_activation_fn": "silu", "use_learned_positional_embeddings": true, "use_rotary_positional_embeddings": true }