TurkishCodeMan commited on
Commit
29ce83e
·
verified ·
1 Parent(s): a39f602

Upload configuration_neurovoice.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. configuration_neurovoice.py +80 -0
configuration_neurovoice.py ADDED
@@ -0,0 +1,80 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import torch
3
+ from transformers import PretrainedConfig, AutoConfig
4
+
5
+
6
+ class NeuroVoiceConfig(PretrainedConfig):
7
+ """
8
+ Configuration class for NeuroVoice-0.5B Neural Audio Language Model.
9
+ Couples a Qwen2.5-0.5B backbone with Kyutai Mimi 24kHz audio codec.
10
+ """
11
+ model_type = "neurovoice"
12
+
13
+ def __init__(
14
+ self,
15
+ text_vocab_size: int = 151936,
16
+ audio_vocab_size: int = 2048,
17
+ num_codebooks: int = 8,
18
+ d_model: int = 896,
19
+ num_heads: int = 14,
20
+ num_kv_heads: int = 2,
21
+ num_layers: int = 24,
22
+ num_depth_layers: int = 4,
23
+ d_ff: int = 4864,
24
+ max_seq_len: int = 2048,
25
+ rope_theta: float = 1000000.0,
26
+ dropout_rate: float = 0.0,
27
+ activation: str = "swiglu",
28
+ dtype: str = "bfloat16",
29
+ use_qk_norm: bool = False,
30
+ pad_token_id: int = 0,
31
+ bos_token_id: int = 1,
32
+ eos_token_id: int = 2,
33
+ unk_token_id: int = 3,
34
+ instruct_token_id: int = 4,
35
+ text_token_id: int = 5,
36
+ ref_audio_token_id: int = 6,
37
+ audio_start_token_id: int = 7,
38
+ audio_end_token_id: int = 8,
39
+ **kwargs,
40
+ ):
41
+ self.text_vocab_size = text_vocab_size
42
+ self.audio_vocab_size = audio_vocab_size
43
+ self.num_codebooks = num_codebooks
44
+ self.d_model = d_model
45
+ self.num_heads = num_heads
46
+ self.num_kv_heads = num_kv_heads
47
+ self.num_layers = num_layers
48
+ self.num_depth_layers = num_depth_layers
49
+ self.d_ff = d_ff
50
+ self.max_seq_len = max_seq_len
51
+ self.rope_theta = rope_theta
52
+ self.dropout_rate = dropout_rate
53
+ self.activation = activation
54
+ self.dtype = dtype
55
+ self.use_qk_norm = use_qk_norm
56
+
57
+ self.instruct_token_id = instruct_token_id
58
+ self.text_token_id = text_token_id
59
+ self.ref_audio_token_id = ref_audio_token_id
60
+ self.audio_start_token_id = audio_start_token_id
61
+ self.audio_end_token_id = audio_end_token_id
62
+
63
+ super().__init__(
64
+ pad_token_id=pad_token_id,
65
+ bos_token_id=bos_token_id,
66
+ eos_token_id=eos_token_id,
67
+ unk_token_id=unk_token_id,
68
+ **kwargs,
69
+ )
70
+
71
+ @property
72
+ def torch_dtype(self) -> torch.dtype:
73
+ if self.dtype == "bfloat16":
74
+ return torch.bfloat16
75
+ elif self.dtype == "float16":
76
+ return torch.float16
77
+ return torch.float32
78
+
79
+
80
+ AutoConfig.register("neurovoice", NeuroVoiceConfig)