{ "motion_tokenizer_config": { "in_channels": 3, "dim": 160, "z_channels": 48, "vae_arch": "wan2.2_vae38", "dim_mult": [ 1, 2, 4, 4 ], "num_res_blocks": 2, "attn_scales": [], "temperal_downsample": [ false, true, true ], "dropout": 0.0, "embedding_dim": 8, "quantizer": "fsq", "levels": [ 8, 5, 5, 5, 5, 5 ], "num_embeddings": null, "beta": 0.25, "persistent_quantizer": false, "act_embedding_num": 32, "qformer_type": "QFormerAdjacentFSingleQ", "qformer_depth": 2, "qformer_num_heads": 4, "connector_out_channels": 1024, "name": "WanMotionHybridTokenizer" }, "cross_attention_dim": 3072, "projector_hidden_dim": 3072, "max_motion_tokens": null }