hf-transformers-bot's picture
Update tiny models for PeAudioVideoEncoder
5f647cc verified
Raw
History Blame Contribute Delete
3.31 kB
{
"architectures": [
"PeAudioVideoEncoder"
],
"attention_bias": false,
"attention_dropout": 0.0,
"audio_config": {
"attention_bias": false,
"attention_dropout": 0.0,
"dac_config": {
"_name_or_path": "",
"architectures": null,
"chunk_size_feed_forward": 0,
"codebook_dim": 32,
"codebook_loss_weight": 1.0,
"codebook_size": 512,
"commitment_loss_weight": 0.25,
"decoder_hidden_size": 16,
"downsampling_ratios": [
2,
4,
4
],
"dtype": null,
"encoder_hidden_size": 16,
"hidden_size": 128,
"hop_length": 32,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"is_encoder_decoder": false,
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"model_type": "dac",
"n_codebooks": 6,
"output_attentions": false,
"output_hidden_states": false,
"problem_type": null,
"quantizer_dropout": 0.0,
"return_dict": true,
"sampling_rate": 16000,
"upsampling_ratios": [
4,
4,
2
]
},
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 32,
"initializer_range": 0.02,
"intermediate_size": 37,
"max_position_embeddings": 512,
"max_window_layers": 28,
"model_type": "pe_audio_encoder",
"num_attention_heads": 2,
"num_hidden_layers": 2,
"num_key_value_heads": 2,
"rms_norm_eps": 1e-05,
"rope_parameters": {
"rope_theta": 20000,
"rope_type": "default"
}
},
"dtype": "float32",
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 32,
"initializer_range": 0.02,
"intermediate_size": 37,
"max_position_embeddings": 512,
"max_window_layers": 28,
"model_type": "pe_audio_video_encoder",
"num_attention_heads": 2,
"num_hidden_layers": 2,
"num_key_value_heads": 2,
"rms_norm_eps": 1e-05,
"rope_parameters": {
"rope_theta": 20000,
"rope_type": "default"
},
"transformers_version": "5.16.0.dev0",
"video_config": {
"attention_bias": false,
"attention_dropout": 0.0,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 32,
"initializer_range": 0.02,
"intermediate_size": 37,
"max_position_embeddings": 512,
"max_window_layers": 28,
"model_type": "pe_video_encoder",
"num_attention_heads": 2,
"num_hidden_layers": 2,
"num_key_value_heads": 2,
"rms_norm_eps": 1e-05,
"rope_parameters": {
"rope_theta": 20000,
"rope_type": "default"
},
"vision_config": {
"_name_or_path": "",
"architecture": "vit_pe_core_large_patch14_336",
"architectures": null,
"chunk_size_feed_forward": 0,
"do_pooling": true,
"dtype": null,
"global_pool": "map",
"initializer_range": 0.02,
"is_encoder_decoder": false,
"label_names": [
"LABEL_0",
"LABEL_1",
"LABEL_2",
"LABEL_3"
],
"model_args": {
"depth": 2,
"embed_dim": 64,
"img_size": [
14,
14
]
},
"model_type": "timm_wrapper",
"num_classes": 4,
"output_attentions": false,
"output_hidden_states": false,
"problem_type": null,
"return_dict": true
}
}
}