Text-to-Audio
MLX
Diffusers
Safetensors
minimax_music3
apple-silicon
macos
minimax
minimax-music3
audio-generation
generative-audio
music-generation
text-to-music
local-inference
Instructions to use appautomaton/MiniMax-Music3-MLX with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use appautomaton/MiniMax-Music3-MLX with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] huggingface-cli download --local-dir MiniMax-Music3-MLX appautomaton/MiniMax-Music3-MLX
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Add files using upload-large-folder tool
Browse files- condition_encoder/config.json +0 -2
- manifest.json +10 -10
- rvq_depth_decoder/config.json +0 -2
- scheduler/scheduler_config.json +0 -2
- transformer/config.json +0 -2
- vocoder/config.json +0 -2
condition_encoder/config.json
CHANGED
|
@@ -1,6 +1,4 @@
|
|
| 1 |
{
|
| 2 |
-
"_class_name": "MiniMaxMusic3ConditionEncoder",
|
| 3 |
-
"_diffusers_version": "0.40.0.dev0",
|
| 4 |
"condition_hidden_dim": 4096,
|
| 5 |
"input_hop_length": 960,
|
| 6 |
"input_sampling_rate": 24000,
|
|
|
|
| 1 |
{
|
|
|
|
|
|
|
| 2 |
"condition_hidden_dim": 4096,
|
| 3 |
"input_hop_length": 960,
|
| 4 |
"input_sampling_rate": 24000,
|
manifest.json
CHANGED
|
@@ -5,8 +5,8 @@
|
|
| 5 |
{
|
| 6 |
"dtypes": [],
|
| 7 |
"path": "condition_encoder/config.json",
|
| 8 |
-
"sha256": "
|
| 9 |
-
"size":
|
| 10 |
"source_path": "condition_encoder/config.json",
|
| 11 |
"source_sha256": "1cdea3f36506719128ae4a29cdfff74b1134e607eb4a504ce6d810568fac8263",
|
| 12 |
"tensor_count": 0
|
|
@@ -126,8 +126,8 @@
|
|
| 126 |
{
|
| 127 |
"dtypes": [],
|
| 128 |
"path": "rvq_depth_decoder/config.json",
|
| 129 |
-
"sha256": "
|
| 130 |
-
"size":
|
| 131 |
"source_path": "rvq_depth_decoder/config.json",
|
| 132 |
"source_sha256": "b38c7f72d4b2fa8667442071f9e4a63ca5a7b5d0859e5e64ee7f79c4bc64520c",
|
| 133 |
"tensor_count": 0
|
|
@@ -150,8 +150,8 @@
|
|
| 150 |
{
|
| 151 |
"dtypes": [],
|
| 152 |
"path": "scheduler/scheduler_config.json",
|
| 153 |
-
"sha256": "
|
| 154 |
-
"size":
|
| 155 |
"source_path": "scheduler/scheduler_config.json",
|
| 156 |
"source_sha256": "b544466124aab333da4874417da0168610067e9c9ff95fcb63d749e4bf891d99",
|
| 157 |
"tensor_count": 0
|
|
@@ -194,8 +194,8 @@
|
|
| 194 |
{
|
| 195 |
"dtypes": [],
|
| 196 |
"path": "transformer/config.json",
|
| 197 |
-
"sha256": "
|
| 198 |
-
"size":
|
| 199 |
"source_path": "transformer/config.json",
|
| 200 |
"source_sha256": "f60856be934127a8b8223510aff2e38a9efd8bbdbc230c3d3339b96e2cb12f00",
|
| 201 |
"tensor_count": 0
|
|
@@ -238,8 +238,8 @@
|
|
| 238 |
{
|
| 239 |
"dtypes": [],
|
| 240 |
"path": "vocoder/config.json",
|
| 241 |
-
"sha256": "
|
| 242 |
-
"size":
|
| 243 |
"source_path": "vocoder/config.json",
|
| 244 |
"source_sha256": "4161afd07ba7d5b949e29c8a7a005592259fba73e9110b6ece9dfd38d3489eac",
|
| 245 |
"tensor_count": 0
|
|
|
|
| 5 |
{
|
| 6 |
"dtypes": [],
|
| 7 |
"path": "condition_encoder/config.json",
|
| 8 |
+
"sha256": "611ba86f32561013f9beb20bc180d0afd671e27906be996830f0bf179886a209",
|
| 9 |
+
"size": 203,
|
| 10 |
"source_path": "condition_encoder/config.json",
|
| 11 |
"source_sha256": "1cdea3f36506719128ae4a29cdfff74b1134e607eb4a504ce6d810568fac8263",
|
| 12 |
"tensor_count": 0
|
|
|
|
| 126 |
{
|
| 127 |
"dtypes": [],
|
| 128 |
"path": "rvq_depth_decoder/config.json",
|
| 129 |
+
"sha256": "2c34c93c0ad15699f96e24f2a70d78924bd58cae7942b64b04d8a0e6f3978cd3",
|
| 130 |
+
"size": 186,
|
| 131 |
"source_path": "rvq_depth_decoder/config.json",
|
| 132 |
"source_sha256": "b38c7f72d4b2fa8667442071f9e4a63ca5a7b5d0859e5e64ee7f79c4bc64520c",
|
| 133 |
"tensor_count": 0
|
|
|
|
| 150 |
{
|
| 151 |
"dtypes": [],
|
| 152 |
"path": "scheduler/scheduler_config.json",
|
| 153 |
+
"sha256": "01eb28ee93a8283713ae2dde5f4e1e008441ccb6ffa4aca2637f3da02966b6b7",
|
| 154 |
+
"size": 392,
|
| 155 |
"source_path": "scheduler/scheduler_config.json",
|
| 156 |
"source_sha256": "b544466124aab333da4874417da0168610067e9c9ff95fcb63d749e4bf891d99",
|
| 157 |
"tensor_count": 0
|
|
|
|
| 194 |
{
|
| 195 |
"dtypes": [],
|
| 196 |
"path": "transformer/config.json",
|
| 197 |
+
"sha256": "2bdb9ba2ace20c423bb6ddc5bfb1de242ce6544e515b926e20bf462fa211b9dc",
|
| 198 |
+
"size": 203,
|
| 199 |
"source_path": "transformer/config.json",
|
| 200 |
"source_sha256": "f60856be934127a8b8223510aff2e38a9efd8bbdbc230c3d3339b96e2cb12f00",
|
| 201 |
"tensor_count": 0
|
|
|
|
| 238 |
{
|
| 239 |
"dtypes": [],
|
| 240 |
"path": "vocoder/config.json",
|
| 241 |
+
"sha256": "6d6baefe31c62cc7514eb81300c76f7006c0e3f92069fd64782096ef6356084c",
|
| 242 |
+
"size": 171,
|
| 243 |
"source_path": "vocoder/config.json",
|
| 244 |
"source_sha256": "4161afd07ba7d5b949e29c8a7a005592259fba73e9110b6ece9dfd38d3489eac",
|
| 245 |
"tensor_count": 0
|
rvq_depth_decoder/config.json
CHANGED
|
@@ -1,6 +1,4 @@
|
|
| 1 |
{
|
| 2 |
-
"_class_name": "MiniMaxMusic3RVQDepthDecoder",
|
| 3 |
-
"_diffusers_version": "0.40.0.dev0",
|
| 4 |
"audio_vocab_size": 1024,
|
| 5 |
"hidden_size": 4096,
|
| 6 |
"intermediate_size": 6144,
|
|
|
|
| 1 |
{
|
|
|
|
|
|
|
| 2 |
"audio_vocab_size": 1024,
|
| 3 |
"hidden_size": 4096,
|
| 4 |
"intermediate_size": 6144,
|
scheduler/scheduler_config.json
CHANGED
|
@@ -1,6 +1,4 @@
|
|
| 1 |
{
|
| 2 |
-
"_class_name": "FlowMatchEulerDiscreteScheduler",
|
| 3 |
-
"_diffusers_version": "0.40.0.dev0",
|
| 4 |
"base_image_seq_len": 256,
|
| 5 |
"base_shift": 0.5,
|
| 6 |
"invert_sigmas": true,
|
|
|
|
| 1 |
{
|
|
|
|
|
|
|
| 2 |
"base_image_seq_len": 256,
|
| 3 |
"base_shift": 0.5,
|
| 4 |
"invert_sigmas": true,
|
transformer/config.json
CHANGED
|
@@ -1,6 +1,4 @@
|
|
| 1 |
{
|
| 2 |
-
"_class_name": "MiniMaxMusic3Transformer1DModel",
|
| 3 |
-
"_diffusers_version": "0.40.0.dev0",
|
| 4 |
"attention_head_dim": 64,
|
| 5 |
"condition_dim": 2048,
|
| 6 |
"ff_inner_dim": 8192,
|
|
|
|
| 1 |
{
|
|
|
|
|
|
|
| 2 |
"attention_head_dim": 64,
|
| 3 |
"condition_dim": 2048,
|
| 4 |
"ff_inner_dim": 8192,
|
vocoder/config.json
CHANGED
|
@@ -1,6 +1,4 @@
|
|
| 1 |
{
|
| 2 |
-
"_class_name": "MiniMaxMusic3Vocoder",
|
| 3 |
-
"_diffusers_version": "0.40.0.dev0",
|
| 4 |
"decoder_hidden_dim": 1536,
|
| 5 |
"decoder_input_dim": 1024,
|
| 6 |
"latent_channels": 128,
|
|
|
|
| 1 |
{
|
|
|
|
|
|
|
| 2 |
"decoder_hidden_dim": 1536,
|
| 3 |
"decoder_input_dim": 1024,
|
| 4 |
"latent_channels": 128,
|