{ "FFE_clip_dim": 1280, "FFE_heads": 32, "FFE_input_fuse_dim": 4096, "FFE_input_img_dim": 2048, "FFE_layers": 10, "FFE_num_queries": 32, "FFE_output_dim": 2048, "FFE_query_dim": 2048, "FFE_time_channel": 3072, "FFE_time_embed_dim": 2048, "FFE_timestep_activation_fn": "silu", "FFE_vit_dim": 1024, "FT_clip_dim": 1280, "FT_depth": 10, "FT_dim_head": 64, "FT_ff_mult": 4, "FT_num_heads": 64, "FT_num_querie": 32, "FT_num_scale": 5, "FT_output_dim": 4096, "FT_txt_dim": 4096, "FT_vit_dim": 1024, "LFE_depth": 10, "LFE_dim_head": 64, "LFE_ff_mult": 4, "LFE_id_dim": 1280, "LFE_num_heads": 16, "LFE_num_id_token": 5, "LFE_num_querie": 32, "LFE_num_scale": 5, "LFE_output_dim": 2048, "LFE_vit_dim": 1024, "_class_name": "ProteusIDTransformer3DModel", "_diffusers_version": "0.33.1", "activation_fn": "gelu-approximate", "attention_bias": true, "attention_head_dim": 64, "cross_attn_dim_head": 128, "cross_attn_interval": 2, "cross_attn_num_heads": 16, "dropout": 0.0, "flip_sin_to_cos": true, "freq_shift": 0, "in_channels": 32, "is_fuse_token": true, "is_fuse_token_v2": true, "is_kps": false, "is_train_face": true, "local_face_scale": 1.0, "max_text_seq_length": 226, "norm_elementwise_affine": true, "norm_eps": 1e-05, "num_attention_heads": 48, "num_layers": 42, "ofs_embed_dim": null, "out_channels": 16, "patch_bias": true, "patch_size": 2, "patch_size_t": null, "sample_frames": 49, "sample_height": 60, "sample_width": 90, "spatial_interpolation_scale": 1.875, "temporal_compression_ratio": 4, "temporal_interpolation_scale": 1.0, "text_embed_dim": 4096, "time_embed_dim": 512, "timestep_activation_fn": "silu", "use_learned_positional_embeddings": true, "use_rotary_positional_embeddings": true }