File size: 6,297 Bytes
3c23f28
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
{
    "model_type": "diffusion_cond",
    "sample_size": 2097152,
    "sample_rate": 44100,
    "audio_channels": 2,
    "model": {
        "pretransform": {
            "type": "autoencoder",
            "iterate_batch": true,
            "config": {
                "encoder": {
                    "type": "oobleck",
                    "requires_grad": false,
                    "config": {
                        "in_channels": 2,
                        "channels": 128,
                        "c_mults": [
                            1,
                            2,
                            4,
                            8,
                            16
                        ],
                        "strides": [
                            2,
                            4,
                            4,
                            8,
                            8
                        ],
                        "latent_dim": 128,
                        "use_snake": true
                    }
                },
                "decoder": {
                    "type": "oobleck",
                    "config": {
                        "out_channels": 2,
                        "channels": 128,
                        "c_mults": [
                            1,
                            2,
                            4,
                            8,
                            16
                        ],
                        "strides": [
                            2,
                            4,
                            4,
                            8,
                            8
                        ],
                        "latent_dim": 64,
                        "use_snake": true,
                        "final_tanh": false
                    }
                },
                "bottleneck": {
                    "type": "vae"
                },
                "latent_dim": 64,
                "downsampling_ratio": 2048,
                "io_channels": 2
            }
        },
        "conditioning": {
            "configs": [
                {
                    "id": "prompt",
                    "type": "t5",
                    "config": {
                        "t5_model_name": "t5-base",
                        "max_length": 128
                    }
                },
                {
                    "id": "seconds_start",
                    "type": "number",
                    "config": {
                        "min_val": 0,
                        "max_val": 512
                    }
                },
                {
                    "id": "seconds_total",
                    "type": "number",
                    "config": {
                        "min_val": 0,
                        "max_val": 512
                    }
                }
            ],
            "cond_dim": 768
        },
        "diffusion": {
            "cross_attention_cond_ids": [
                "prompt",
                "seconds_start",
                "seconds_total"
            ],
            "global_cond_ids": [
                "seconds_start",
                "seconds_total"
            ],
            "type": "dit",
            "config": {
                "io_channels": 64,
                "embed_dim": 1536,
                "depth": 24,
                "num_heads": 24,
                "cond_token_dim": 768,
                "global_cond_dim": 1536,
                "project_cond_tokens": false,
                "transformer_type": "continuous_transformer"
            }
        },
        "io_channels": 64
    },
    "training": {
        "use_ema": true,
        "log_loss_info": false,
        "optimizer_configs": {
            "diffusion": {
                "optimizer": {
                    "type": "AdamW",
                    "config": {
                        "lr": 5e-5,
                        "betas": [
                            0.9,
                            0.999
                        ],
                        "weight_decay": 1e-3
                    }
                },
                "scheduler": {
                    "type": "InverseLR",
                    "config": {
                        "inv_gamma": 1000000,
                        "power": 0.5,
                        "warmup": 0.99
                    }
                }
            }
        },
        "demo": {
            "demo_every": 687,
            "demo_steps": 250,
            "num_demos": 4,
            "demo_cond": [
                {
                    "prompt": "Format: Solo | Genre: Trap | Sub-Genre: Cinematic Piano / Ambient | Instruments: Acoustic Piano | Moods: Sad, Melancholic, Reflective, Somber, Introspective | Styles: Expressive, Emotional, Sparse | Tempo: Slow | BPM: 70 | Key: Am",
                    "seconds_start": 0,
                    "seconds_total": 15
                },
                {
                    "prompt": "Format: Solo | Genre: Trap | Sub-Genre: Melodic Trap / Future Pop | Instruments: Synth Lead, Synth Plucks, Atmospheric Pads | Moods: Bright, Uplifting, Carefree, Energetic | Styles: Wavy, Catchy, Modern, Smooth | Tempo: Fast | BPM: 155 | Key: G",
                    "seconds_start": 0,
                    "seconds_total": 30
                },
                {
                    "prompt": "Format: Solo | Genre: Trap | Sub-Genre: Hard Trap / Cinematic Trap | Instruments: Synth Pads (dark & evolving), Sub Bass, FX Risers, Distorted Synth Stabs | Moods: Epic, Deep, Intense, Dark, Anticipatory, Powerful | Styles: Atmospheric, Building, Menacing, Cinematic | Tempo: Medium-Fast | BPM: 150 | Key: Fm",
                    "seconds_start": 0,
                    "seconds_total": 12
                },
                {
                    "prompt": "Format: Solo | Genre: Trap | Sub-Genre: Trap | Instruments: 808 Bass, Programmed Drums, Hi-Hats, Snare | Moods: Dark, Menacing, Energetic, Driving | Styles: Hard-hitting, Rhythmic, Urban, Gritty | Tempo: Mid-Fast, Driving | BPM: 130 | Key: Cm",
                    "seconds_start": 0,
                    "seconds_total": 18
                }
            ],
            "demo_cfg_scales": [
                7,
                9,
                10
            ]
        }
    }
}