Essid commited on
Commit
1c1eccd
·
verified ·
1 Parent(s): 555f099

Upload 2 files

Browse files
config_vocals_mdx23c.yaml ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ audio:
2
+ chunk_size: 261120
3
+ dim_f: 4096
4
+ dim_t: 256
5
+ hop_length: 1024
6
+ n_fft: 8192
7
+ num_channels: 2
8
+ sample_rate: 44100
9
+ min_mean_abs: 0.001
10
+
11
+ model:
12
+ act: gelu
13
+ bottleneck_factor: 4
14
+ growth: 128
15
+ norm: InstanceNorm
16
+ num_blocks_per_scale: 2
17
+ num_channels: 128
18
+ num_scales: 5
19
+ num_subbands: 4
20
+ scale:
21
+ - 2
22
+ - 2
23
+
24
+ training:
25
+ batch_size: 2
26
+ gradient_accumulation_steps: 1
27
+ grad_clip: 0
28
+ instruments:
29
+ - vocals
30
+ - other
31
+ lr: 1.0e-05
32
+ patience: 2
33
+ reduce_factor: 0.95
34
+ target_instrument: null
35
+ num_epochs: 1000
36
+ num_steps: 1000
37
+ q: 0.95
38
+ coarse_loss_clip: true
39
+ ema_momentum: 0.999
40
+ optimizer: adam
41
+ read_metadata_procs: 8 # Number of processes to use during metadata reading for dataset. Can speed up metadata generation
42
+ other_fix: true # it's needed for checking on multisong dataset if other is actually instrumental
43
+ use_amp: true # enable or disable usage of mixed precision (float16) - usually it must be true
44
+
45
+ augmentations:
46
+ enable: false # enable or disable all augmentations (to fast disable if needed)
47
+ loudness: true # randomly change loudness of each stem on the range (loudness_min; loudness_max)
48
+ loudness_min: 0.5
49
+ loudness_max: 1.5
50
+ mixup: true # mix several stems of same type with some probability (only works for dataset types: 1, 2, 3)
51
+ mixup_probs: !!python/tuple # 2 additional stems of the same type (1st with prob 0.2, 2nd with prob 0.02)
52
+ - 0.2
53
+ - 0.02
54
+ mixup_loudness_min: 0.5
55
+ mixup_loudness_max: 1.5
56
+
57
+ # apply mp3 compression to mixture only (emulate downloading mp3 from internet)
58
+ mp3_compression_on_mixture: 0.01
59
+ mp3_compression_on_mixture_bitrate_min: 32
60
+ mp3_compression_on_mixture_bitrate_max: 320
61
+ mp3_compression_on_mixture_backend: "lameenc"
62
+
63
+ all:
64
+ channel_shuffle: 0.5 # Set 0 or lower to disable
65
+ random_inverse: 0.1 # inverse track (better lower probability)
66
+ random_polarity: 0.5 # polarity change (multiply waveform to -1)
67
+ mp3_compression: 0.01
68
+ mp3_compression_min_bitrate: 32
69
+ mp3_compression_max_bitrate: 320
70
+ mp3_compression_backend: "lameenc"
71
+
72
+ vocals:
73
+ pitch_shift: 0.1
74
+ pitch_shift_min_semitones: -5
75
+ pitch_shift_max_semitones: 5
76
+ seven_band_parametric_eq: 0.25
77
+ seven_band_parametric_eq_min_gain_db: -9
78
+ seven_band_parametric_eq_max_gain_db: 9
79
+ tanh_distortion: 0.1
80
+ tanh_distortion_min: 0.1
81
+ tanh_distortion_max: 0.7
82
+ other:
83
+ pitch_shift: 0.1
84
+ pitch_shift_min_semitones: -4
85
+ pitch_shift_max_semitones: 4
86
+ gaussian_noise: 0.1
87
+ gaussian_noise_min_amplitude: 0.001
88
+ gaussian_noise_max_amplitude: 0.015
89
+ time_stretch: 0.01
90
+ time_stretch_min_rate: 0.8
91
+ time_stretch_max_rate: 1.25
92
+
93
+ inference:
94
+ batch_size: 1
95
+ dim_t: 256
96
+ num_overlap: 4
model_mdx23c_ep_0_sdr_14.6632.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f75a7e9cf05664fd265e14c7225d85c15c6498b98c55a7ddbfd44fbd0ec13841
3
+ size 448103606