Haopeng Gen commited on Oct 6, 2024

Commit

b348852

1 Parent(s): a22f4e8

add vocoders

Files changed (19) hide show

.gitattributes +7 -0
README.md +3 -3
hifigan.16k_320/checkpoint-400000steps.pkl +3 -0
hifigan.16k_320/config.yml +191 -0
hifigan.16k_320/stats.h5 +3 -0
hifigan_hubert.16k_320/checkpoint-400000steps.pkl +3 -0
hifigan_hubert.16k_320/config.yml +195 -0
hifigan_hubert_unit_km500.16k_320/checkpoint-800000steps.pkl +3 -0
hifigan_hubert_unit_km500.16k_320/config.yml +195 -0
hifigan_hubert_unit_km500.16k_320/hifigan_hubert.v1.yaml +176 -0
ppg_sxliu_decoder_V006/checkpoint-38000steps.pkl +3 -0
ppg_sxliu_decoder_V006/config.yml +61 -0
ppg_sxliu_decoder_V006/stats.h5 +3 -0
pwg.16k_256/checkpoint-400000steps.pkl +3 -0
pwg.16k_256/config.yml +104 -0
pwg.16k_256/stats.h5 +3 -0
s3prl-vc-ppg_sxliu/checkpoint-50000steps.pkl +1 -0
s3prl-vc-ppg_sxliu/config.yml +61 -0
s3prl-vc-ppg_sxliu/stats.h5 +3 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,10 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+s3prl-vc-ppg_sxliu filter=lfs diff=lfs merge=lfs -text
+hifigan.16k_320 filter=lfs diff=lfs merge=lfs -text
+hifigan_hubert.16k_320 filter=lfs diff=lfs merge=lfs -text
+hifigan_hubert_unit_km500.16k_320 filter=lfs diff=lfs merge=lfs -text
+ppg_sxliu_decoder_V006 filter=lfs diff=lfs merge=lfs -text
+pwg.16k_256 filter=lfs diff=lfs merge=lfs -text
+README.md filter=lfs diff=lfs merge=lfs -text

README.md CHANGED Viewed

@@ -1,3 +1,3 @@
----
-license: apache-2.0
----

+version https://git-lfs.github.com/spec/v1
+oid sha256:4bcf87ecfbbb8e07a01b21415a970c8b53a5283bf6872b657040d3f45c9241f7
+size 31

hifigan.16k_320/checkpoint-400000steps.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:99844ee49f8a011ad9a245219c19cd6a10d751539198b24e64929baf0d8c933e
+size 1119163385

hifigan.16k_320/config.yml ADDED Viewed

	@@ -0,0 +1,191 @@

+allow_cache: true
+batch_max_steps: 10240
+batch_size: 16
+config: conf/hifigan.16k_320.yaml
+dev_dumpdir: dump/dev/norm
+dev_feats_scp: null
+dev_segments: null
+dev_wav_scp: null
+discriminator_adv_loss_params:
+  average_by_discriminators: false
+discriminator_grad_norm: -1
+discriminator_optimizer_params:
+  betas:
+  - 0.5
+  - 0.9
+  lr: 0.0002
+  weight_decay: 0.0
+discriminator_optimizer_type: Adam
+discriminator_params:
+  follow_official_norm: true
+  period_discriminator_params:
+    bias: true
+    channels: 32
+    downsample_scales:
+    - 3
+    - 3
+    - 3
+    - 3
+    - 1
+    in_channels: 1
+    kernel_sizes:
+    - 5
+    - 3
+    max_downsample_channels: 1024
+    nonlinear_activation: LeakyReLU
+    nonlinear_activation_params:
+      negative_slope: 0.1
+    out_channels: 1
+    use_spectral_norm: false
+    use_weight_norm: true
+  periods:
+  - 2
+  - 3
+  - 5
+  - 7
+  - 11
+  scale_discriminator_params:
+    bias: true
+    channels: 128
+    downsample_scales:
+    - 4
+    - 4
+    - 4
+    - 4
+    - 1
+    in_channels: 1
+    kernel_sizes:
+    - 15
+    - 41
+    - 5
+    - 3
+    max_downsample_channels: 1024
+    max_groups: 16
+    nonlinear_activation: LeakyReLU
+    nonlinear_activation_params:
+      negative_slope: 0.1
+    out_channels: 1
+  scale_downsample_pooling: AvgPool1d
+  scale_downsample_pooling_params:
+    kernel_size: 4
+    padding: 2
+    stride: 2
+  scales: 3
+discriminator_scheduler_params:
+  gamma: 0.5
+  milestones:
+  - 200000
+  - 400000
+  - 600000
+  - 800000
+discriminator_scheduler_type: MultiStepLR
+discriminator_train_start_steps: 0
+discriminator_type: HiFiGANMultiScaleMultiPeriodDiscriminator
+distributed: false
+eval_interval_steps: 1000
+feat_match_loss_params:
+  average_by_discriminators: false
+  average_by_layers: false
+  include_final_outputs: false
+fft_size: 1280
+fmax: 7600
+fmin: 80
+format: hdf5
+generator_adv_loss_params:
+  average_by_discriminators: false
+generator_grad_norm: -1
+generator_optimizer_params:
+  betas:
+  - 0.5
+  - 0.9
+  lr: 0.0002
+  weight_decay: 0.0
+generator_optimizer_type: Adam
+generator_params:
+  bias: true
+  channels: 640
+  in_channels: 80
+  kernel_size: 7
+  nonlinear_activation: LeakyReLU
+  nonlinear_activation_params:
+    negative_slope: 0.1
+  out_channels: 1
+  resblock_dilations:
+  - - 1
+    - 3
+    - 5
+  - - 1
+    - 3
+    - 5
+  - - 1
+    - 3
+    - 5
+  resblock_kernel_sizes:
+  - 3
+  - 7
+  - 11
+  upsample_kernel_sizes:
+  - 20
+  - 16
+  - 4
+  - 4
+  upsample_scales:
+  - 10
+  - 8
+  - 2
+  - 2
+  use_additional_convs: true
+  use_weight_norm: true
+generator_scheduler_params:
+  gamma: 0.5
+  milestones:
+  - 200000
+  - 400000
+  - 600000
+  - 800000
+generator_scheduler_type: MultiStepLR
+generator_train_start_steps: 1
+generator_type: HiFiGANGenerator
+global_gain_scale: 1.0
+hop_size: 320
+lambda_adv: 1.0
+lambda_aux: 45.0
+lambda_feat_match: 2.0
+log_interval_steps: 100
+mel_loss_params:
+  fft_size: 1280
+  fmax: 8000
+  fmin: 0
+  fs: 16000
+  hop_size: 320
+  log_base: null
+  num_mels: 80
+  win_length: null
+  window: hann
+num_mels: 80
+num_save_intermediate_results: 4
+num_workers: 2
+outdir: exp/train_nodev_hifigan.16k_320
+pin_memory: true
+pretrain: ''
+rank: 0
+remove_short_samples: false
+resume: ''
+sampling_rate: 16000
+save_interval_steps: 10000
+train_dumpdir: dump/train_nodev/norm
+train_feats_scp: null
+train_max_steps: 400000
+train_segments: null
+train_wav_scp: null
+trim_frame_size: 1024
+trim_hop_size: 320
+trim_silence: false
+trim_threshold_in_db: 20
+use_feat_match_loss: true
+use_mel_loss: true
+use_stft_loss: false
+verbose: 1
+version: 0.6.2a
+win_length: null
+window: hann

hifigan.16k_320/stats.h5 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:acdf123b29e8e9d857006144b46583da550af45dd865b89f2f609a45a80eee48
+size 4912

hifigan_hubert.16k_320/checkpoint-400000steps.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3bf20a87037cb70e3151309135b73890927f4174daec3f92d5ea7312a6095c6d
+size 1042691825

hifigan_hubert.16k_320/config.yml ADDED Viewed

	@@ -0,0 +1,195 @@

+allow_cache: true
+batch_max_steps: 10240
+batch_size: 32
+config: ./conf/hifigan_hubert.v1.yaml
+dev_dumpdir: dump/V006_SS_max_valid_dev/raw
+dev_feats_scp: null
+dev_segments: null
+dev_wav_scp: null
+discriminator_adv_loss_params:
+  average_by_discriminators: false
+discriminator_grad_norm: -1
+discriminator_optimizer_params:
+  betas:
+  - 0.5
+  - 0.9
+  lr: 0.0002
+  weight_decay: 0.0
+discriminator_optimizer_type: Adam
+discriminator_params:
+  follow_official_norm: true
+  period_discriminator_params:
+    bias: true
+    channels: 32
+    downsample_scales:
+    - 3
+    - 3
+    - 3
+    - 3
+    - 1
+    in_channels: 1
+    kernel_sizes:
+    - 5
+    - 3
+    max_downsample_channels: 1024
+    nonlinear_activation: LeakyReLU
+    nonlinear_activation_params:
+      negative_slope: 0.1
+    out_channels: 1
+    use_spectral_norm: false
+    use_weight_norm: true
+  periods:
+  - 2
+  - 3
+  - 5
+  - 7
+  - 11
+  scale_discriminator_params:
+    bias: true
+    channels: 128
+    downsample_scales:
+    - 4
+    - 4
+    - 4
+    - 4
+    - 1
+    in_channels: 1
+    kernel_sizes:
+    - 15
+    - 41
+    - 5
+    - 3
+    max_downsample_channels: 1024
+    max_groups: 16
+    nonlinear_activation: LeakyReLU
+    nonlinear_activation_params:
+      negative_slope: 0.1
+    out_channels: 1
+  scale_downsample_pooling: AvgPool1d
+  scale_downsample_pooling_params:
+    kernel_size: 4
+    padding: 2
+    stride: 2
+  scales: 3
+discriminator_scheduler_params:
+  gamma: 0.5
+  milestones:
+  - 200000
+  - 400000
+  - 600000
+  - 800000
+discriminator_scheduler_type: MultiStepLR
+discriminator_train_start_steps: 0
+discriminator_type: HiFiGANMultiScaleMultiPeriodDiscriminator
+distributed: false
+eval_interval_steps: 1000
+feat_match_loss_params:
+  average_by_discriminators: false
+  average_by_layers: false
+  include_final_outputs: true
+fft_size: null
+fmax: null
+fmin: null
+format: hdf5
+generator_adv_loss_params:
+  average_by_discriminators: false
+generator_grad_norm: -1
+generator_optimizer_params:
+  betas:
+  - 0.5
+  - 0.9
+  lr: 0.0002
+  weight_decay: 0.0
+generator_optimizer_type: Adam
+generator_params:
+  bias: true
+  channels: 512
+  concat_spk_emb: false
+  in_channels: 512
+  kernel_size: 7
+  nonlinear_activation: LeakyReLU
+  nonlinear_activation_params:
+    negative_slope: 0.1
+  num_embs: 100
+  num_spk_embs: 128
+  out_channels: 1
+  resblock_dilations:
+  - - 1
+    - 3
+    - 5
+  - - 1
+    - 3
+    - 5
+  - - 1
+    - 3
+    - 5
+  resblock_kernel_sizes:
+  - 3
+  - 7
+  - 11
+  spk_emb_dim: 512
+  upsample_kernel_sizes:
+  - 20
+  - 16
+  - 4
+  - 4
+  upsample_scales:
+  - 10
+  - 8
+  - 2
+  - 2
+  use_additional_convs: true
+  use_weight_norm: true
+generator_scheduler_params:
+  gamma: 0.5
+  milestones:
+  - 200000
+  - 400000
+  - 600000
+  - 800000
+generator_scheduler_type: MultiStepLR
+generator_train_start_steps: 1
+generator_type: DiscreteSymbolHiFiGANGenerator
+global_gain_scale: 1.0
+hop_size: 320
+lambda_adv: 1.0
+lambda_aux: 45.0
+lambda_feat_match: 2.0
+log_interval_steps: 100
+mel_loss_params:
+  fft_size: 1280
+  fmax: 8000
+  fmin: 0
+  fs: 16000
+  hop_size: 320
+  log_base: null
+  num_mels: 80
+  win_length: null
+  window: hann
+num_mels: 2
+num_save_intermediate_results: 4
+num_workers: 2
+outdir: exp/V006_SS_max_valid_train_2000_vctk_hifigan_hubert.v1
+pin_memory: true
+pretrain: ''
+rank: 0
+remove_short_samples: false
+resume: exp/V006_SS_max_valid_train_2000_vctk_hifigan_hubert.v1/checkpoint-300steps.pkl
+sampling_rate: 16000
+save_interval_steps: 50000
+train_dumpdir: dump/V006_SS_max_valid_train_2000/raw
+train_feats_scp: null
+train_max_steps: 2500000
+train_segments: null
+train_wav_scp: null
+trim_frame_size: 1024
+trim_hop_size: 320
+trim_silence: false
+trim_threshold_in_db: 20
+use_feat_match_loss: true
+use_mel_loss: true
+use_stft_loss: false
+verbose: 1
+version: 0.6.2a
+win_length: null
+window: null

hifigan_hubert_unit_km500.16k_320/checkpoint-800000steps.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c947a9745940990f39799cbee794a3cde8326e2dbc720e7976ccea4675a213d7
+size 1045149425

hifigan_hubert_unit_km500.16k_320/config.yml ADDED Viewed

	@@ -0,0 +1,195 @@

+allow_cache: true
+batch_max_steps: 10240
+batch_size: 32
+config: ./conf/hifigan_hubert.v1.yaml
+dev_dumpdir: dump/V006_SS_max_valid_dev/raw
+dev_feats_scp: null
+dev_segments: null
+dev_wav_scp: null
+discriminator_adv_loss_params:
+  average_by_discriminators: false
+discriminator_grad_norm: -1
+discriminator_optimizer_params:
+  betas:
+  - 0.5
+  - 0.9
+  lr: 0.0002
+  weight_decay: 0.0
+discriminator_optimizer_type: Adam
+discriminator_params:
+  follow_official_norm: true
+  period_discriminator_params:
+    bias: true
+    channels: 32
+    downsample_scales:
+    - 3
+    - 3
+    - 3
+    - 3
+    - 1
+    in_channels: 1
+    kernel_sizes:
+    - 5
+    - 3
+    max_downsample_channels: 1024
+    nonlinear_activation: LeakyReLU
+    nonlinear_activation_params:
+      negative_slope: 0.1
+    out_channels: 1
+    use_spectral_norm: false
+    use_weight_norm: true
+  periods:
+  - 2
+  - 3
+  - 5
+  - 7
+  - 11
+  scale_discriminator_params:
+    bias: true
+    channels: 128
+    downsample_scales:
+    - 4
+    - 4
+    - 4
+    - 4
+    - 1
+    in_channels: 1
+    kernel_sizes:
+    - 15
+    - 41
+    - 5
+    - 3
+    max_downsample_channels: 1024
+    max_groups: 16
+    nonlinear_activation: LeakyReLU
+    nonlinear_activation_params:
+      negative_slope: 0.1
+    out_channels: 1
+  scale_downsample_pooling: AvgPool1d
+  scale_downsample_pooling_params:
+    kernel_size: 4
+    padding: 2
+    stride: 2
+  scales: 3
+discriminator_scheduler_params:
+  gamma: 0.5
+  milestones:
+  - 200000
+  - 400000
+  - 600000
+  - 800000
+discriminator_scheduler_type: MultiStepLR
+discriminator_train_start_steps: 0
+discriminator_type: HiFiGANMultiScaleMultiPeriodDiscriminator
+distributed: false
+eval_interval_steps: 1000
+feat_match_loss_params:
+  average_by_discriminators: false
+  average_by_layers: false
+  include_final_outputs: true
+fft_size: null
+fmax: null
+fmin: null
+format: hdf5
+generator_adv_loss_params:
+  average_by_discriminators: false
+generator_grad_norm: -1
+generator_optimizer_params:
+  betas:
+  - 0.5
+  - 0.9
+  lr: 0.0002
+  weight_decay: 0.0
+generator_optimizer_type: Adam
+generator_params:
+  bias: true
+  channels: 512
+  concat_spk_emb: false
+  in_channels: 512
+  kernel_size: 7
+  nonlinear_activation: LeakyReLU
+  nonlinear_activation_params:
+    negative_slope: 0.1
+  num_embs: 500
+  num_spk_embs: 128
+  out_channels: 1
+  resblock_dilations:
+  - - 1
+    - 3
+    - 5
+  - - 1
+    - 3
+    - 5
+  - - 1
+    - 3
+    - 5
+  resblock_kernel_sizes:
+  - 3
+  - 7
+  - 11
+  spk_emb_dim: 512
+  upsample_kernel_sizes:
+  - 20
+  - 16
+  - 4
+  - 4
+  upsample_scales:
+  - 10
+  - 8
+  - 2
+  - 2
+  use_additional_convs: true
+  use_weight_norm: true
+generator_scheduler_params:
+  gamma: 0.5
+  milestones:
+  - 200000
+  - 400000
+  - 600000
+  - 800000
+generator_scheduler_type: MultiStepLR
+generator_train_start_steps: 1
+generator_type: DiscreteSymbolHiFiGANGenerator
+global_gain_scale: 1.0
+hop_size: 320
+lambda_adv: 1.0
+lambda_aux: 45.0
+lambda_feat_match: 2.0
+log_interval_steps: 100
+mel_loss_params:
+  fft_size: 1024
+  fmax: 8000
+  fmin: 0
+  fs: 16000
+  hop_size: 256
+  log_base: null
+  num_mels: 80
+  win_length: null
+  window: hann
+num_mels: 2
+num_save_intermediate_results: 4
+num_workers: 2
+outdir: exp/V006_SS_max_valid_train_2000_vctk_hifigan_hubert.v1
+pin_memory: true
+pretrain: ''
+rank: 0
+remove_short_samples: false
+resume: exp/V006_SS_max_valid_train_2000_vctk_hifigan_hubert.v1/checkpoint-104steps.pkl
+sampling_rate: 16000
+save_interval_steps: 50000
+train_dumpdir: dump/V006_SS_max_valid_train_2000/raw
+train_feats_scp: null
+train_max_steps: 2500000
+train_segments: null
+train_wav_scp: null
+trim_frame_size: 1024
+trim_hop_size: 256
+trim_silence: false
+trim_threshold_in_db: 20
+use_feat_match_loss: true
+use_mel_loss: true
+use_stft_loss: false
+verbose: 1
+version: 0.6.2a
+win_length: null
+window: null

hifigan_hubert_unit_km500.16k_320/hifigan_hubert.v1.yaml ADDED Viewed

	@@ -0,0 +1,176 @@

+# This configuration is based on HiFiGAN V1, derived
+# from official repository (https://github.com/jik876/hifi-gan).
+###########################################################
+#                FEATURE EXTRACTION SETTING               #
+###########################################################
+sampling_rate: 16000     # Sampling rate.
+fft_size: null           # FFT size.
+hop_size: 320            # Hop size.
+win_length: null         # Window length.
+                         # If set to null, it will be the same as fft_size.
+window: null             # Window function.
+num_mels: 2              # Number of mel basis.
+fmin: null               # Minimum freq in mel basis calculation.
+fmax: null               # Maximum frequency in mel basis calculation.
+global_gain_scale: 1.0   # Will be multiplied to all of waveform.
+trim_silence: false      # Whether to trim the start and end of silence.
+trim_threshold_in_db: 20 # Need to tune carefully if the recording is not good.
+trim_frame_size: 1024    # Frame size in trimming.
+trim_hop_size: 256       # Hop size in trimming.
+format: "hdf5"           # Feature file format. "npy" or "hdf5" is supported.
+###########################################################
+#         GENERATOR NETWORK ARCHITECTURE SETTING          #
+###########################################################
+generator_type: DiscreteSymbolHiFiGANGenerator
+generator_params:
+    in_channels: 512                      # Number of input channels.
+    out_channels: 1                       # Number of output channels.
+    channels: 512                         # Number of initial channels.
+    num_embs: 500
+    num_spk_embs: 128
+    spk_emb_dim: 512
+    concat_spk_emb: false
+    kernel_size: 7                        # Kernel size of initial and final conv layers.
+    upsample_scales: [10, 8, 2, 2]        # Upsampling scales.
+    upsample_kernel_sizes: [20, 16, 4, 4] # Kernel size for upsampling layers.
+    resblock_kernel_sizes: [3, 7, 11]     # Kernel size for residual blocks.
+    resblock_dilations:                   # Dilations for residual blocks.
+        - [1, 3, 5]
+        - [1, 3, 5]
+        - [1, 3, 5]
+    use_additional_convs: true            # Whether to use additional conv layer in residual blocks.
+    bias: true                            # Whether to use bias parameter in conv.
+    nonlinear_activation: "LeakyReLU"     # Nonlinear activation type.
+    nonlinear_activation_params:          # Nonlinear activation paramters.
+        negative_slope: 0.1
+    use_weight_norm: true                 # Whether to apply weight normalization.
+###########################################################
+#       DISCRIMINATOR NETWORK ARCHITECTURE SETTING        #
+###########################################################
+discriminator_type: HiFiGANMultiScaleMultiPeriodDiscriminator
+discriminator_params:
+    scales: 3                              # Number of multi-scale discriminator.
+    scale_downsample_pooling: "AvgPool1d"  # Pooling operation for scale discriminator.
+    scale_downsample_pooling_params:
+        kernel_size: 4                     # Pooling kernel size.
+        stride: 2                          # Pooling stride.
+        padding: 2                         # Padding size.
+    scale_discriminator_params:
+        in_channels: 1                     # Number of input channels.
+        out_channels: 1                    # Number of output channels.
+        kernel_sizes: [15, 41, 5, 3]       # List of kernel sizes.
+        channels: 128                      # Initial number of channels.
+        max_downsample_channels: 1024      # Maximum number of channels in downsampling conv layers.
+        max_groups: 16                     # Maximum number of groups in downsampling conv layers.
+        bias: true
+        downsample_scales: [4, 4, 4, 4, 1] # Downsampling scales.
+        nonlinear_activation: "LeakyReLU"  # Nonlinear activation.
+        nonlinear_activation_params:
+            negative_slope: 0.1
+    follow_official_norm: true             # Whether to follow the official norm setting.
+    periods: [2, 3, 5, 7, 11]              # List of period for multi-period discriminator.
+    period_discriminator_params:
+        in_channels: 1                     # Number of input channels.
+        out_channels: 1                    # Number of output channels.
+        kernel_sizes: [5, 3]               # List of kernel sizes.
+        channels: 32                       # Initial number of channels.
+        downsample_scales: [3, 3, 3, 3, 1] # Downsampling scales.
+        max_downsample_channels: 1024      # Maximum number of channels in downsampling conv layers.
+        bias: true                         # Whether to use bias parameter in conv layer."
+        nonlinear_activation: "LeakyReLU"  # Nonlinear activation.
+        nonlinear_activation_params:       # Nonlinear activation paramters.
+            negative_slope: 0.1
+        use_weight_norm: true              # Whether to apply weight normalization.
+        use_spectral_norm: false           # Whether to apply spectral normalization.
+###########################################################
+#                   STFT LOSS SETTING                     #
+###########################################################
+use_stft_loss: false # Whether to use multi-resolution STFT loss.
+use_mel_loss: true   # Whether to use Mel-spectrogram loss.
+mel_loss_params:     # Mel-spectrogram loss parameters.
+    fs: 16000
+    fft_size: 1024
+    hop_size: 256
+    win_length: null
+    window: "hann"
+    num_mels: 80
+    fmin: 0
+    fmax: 8000
+    log_base: null   # Log base. If set to null, use natural logarithm.
+generator_adv_loss_params:
+    average_by_discriminators: false # Whether to average loss by #discriminators.
+discriminator_adv_loss_params:
+    average_by_discriminators: false # Whether to average loss by #discriminators.
+use_feat_match_loss: true
+feat_match_loss_params:
+    average_by_discriminators: false # Whether to average loss by #discriminators.
+    average_by_layers: false         # Whether to average loss by #layers in each discriminator.
+    include_final_outputs: true      # Whether to include final outputs in feat match loss calculation.
+###########################################################
+#               ADVERSARIAL LOSS SETTING                  #
+###########################################################
+lambda_aux: 45.0       # Loss balancing coefficient for STFT loss.
+lambda_adv: 1.0        # Loss balancing coefficient for adversarial loss.
+lambda_feat_match: 2.0 # Loss balancing coefficient for feat match loss..
+###########################################################
+#                  DATA LOADER SETTING                    #
+###########################################################
+batch_size: 32              # Batch size.
+batch_max_steps: 10240      # Length of each audio in batch. Make sure dividable by hop_size.
+pin_memory: true            # Whether to pin memory in Pytorch DataLoader.
+num_workers: 2              # Number of workers in Pytorch DataLoader.
+remove_short_samples: false # Whether to remove samples the length of which are less than batch_max_steps.
+allow_cache: true           # Whether to allow cache in dataset. If true, it requires cpu memory.
+###########################################################
+#             OPTIMIZER & SCHEDULER SETTING               #
+###########################################################
+generator_optimizer_type: Adam
+generator_optimizer_params:
+    lr: 2.0e-4
+    betas: [0.5, 0.9]
+    weight_decay: 0.0
+generator_scheduler_type: MultiStepLR
+generator_scheduler_params:
+    gamma: 0.5
+    milestones:
+        - 200000
+        - 400000
+        - 600000
+        - 800000
+generator_grad_norm: -1
+discriminator_optimizer_type: Adam
+discriminator_optimizer_params:
+    lr: 2.0e-4
+    betas: [0.5, 0.9]
+    weight_decay: 0.0
+discriminator_scheduler_type: MultiStepLR
+discriminator_scheduler_params:
+    gamma: 0.5
+    milestones:
+        - 200000
+        - 400000
+        - 600000
+        - 800000
+discriminator_grad_norm: -1
+###########################################################
+#                    INTERVAL SETTING                     #
+###########################################################
+generator_train_start_steps: 1     # Number of steps to start to train discriminator.
+discriminator_train_start_steps: 0 # Number of steps to start to train discriminator.
+train_max_steps: 2500000           # Number of training steps.
+save_interval_steps: 50000         # Interval steps to save checkpoint.
+eval_interval_steps: 1000          # Interval steps to evaluate the network.
+log_interval_steps: 100            # Interval steps to record the training log.
+###########################################################
+#                     OTHER SETTING                       #
+###########################################################
+num_save_intermediate_results: 4  # Number of results to be saved as intermediate results.

ppg_sxliu_decoder_V006/checkpoint-38000steps.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e10c9f73d33dd39fe7ef3f52f6fd513fdfa621c933b1abeeb901d5ee0b857af9
+size 339924234

ppg_sxliu_decoder_V006/config.yml ADDED Viewed

	@@ -0,0 +1,61 @@

+additional_config: null
+allow_cache: true
+batch_size: 6
+config: conf/taco2_ar_V006_S1.yaml
+dev_scp: data/V006_S1_max_valid_dev/wav.scp
+dev_spemb_scp: null
+distributed: false
+eval_interval_steps: 1000
+fft_size: 1024
+fmax: 7600
+fmin: 80
+global_gain_scale: 1.0
+grad_norm: 1.0
+hop_size: 256
+init_checkpoint: ''
+log_interval_steps: 100
+main_loss_type: L1Loss
+model_params:
+  ar: true
+  encoder_type: taco2
+  hidden_dim: 1024
+  lstmp_dropout_rate: 0.2
+  lstmp_layernorm: false
+  lstmp_layers: 2
+  lstmp_proj_dim: 256
+  prenet_dim: 256
+  prenet_dropout_rate: 0.5
+  prenet_layers: 2
+model_type: Taco2_AR
+num_mels: 80
+num_save_intermediate_results: 4
+num_workers: 2
+optimizer_params:
+  lr: 0.0001
+optimizer_type: AdamW
+outdir: exp/V006_S1_max_valid_ppg_sxliu_taco2_ar_V006_S1
+pin_memory: true
+rank: 0
+resume: exp/V006_S1_max_valid_ppg_sxliu_taco2_ar_V006_S1/checkpoint-10000steps.pkl
+sampling_rate: 16000
+save_interval_steps: 1000
+scheduler: linear_schedule_with_warmup
+scheduler_params:
+  num_warmup_steps: 4000
+train_max_steps: 100000
+train_scp: data/V006_S1_max_valid_train/wav.scp
+train_spemb_scp: null
+trg_stats: exp/V006_S1_max_valid_ppg_sxliu_taco2_ar_V006_S1/stats.h5
+trim_frame_size: 2048
+trim_hop_size: 512
+trim_silence: false
+trim_threshold_in_db: 60
+upstream: ppg_sxliu
+verbose: 1
+version: 0.3.0
+vocoder:
+    checkpoint: /home/kevingenghaopeng/vocoder/ParallelWaveGAN/egs/V006/voc1/exp/train_nodev_parallel_wavegan.v1/checkpoint-400000steps.pkl
+    config: /home/kevingenghaopeng/vocoder/ParallelWaveGAN/egs/V006/voc1/exp/train_nodev_parallel_wavegan.v1/config.yml
+    stats: /home/kevingenghaopeng/vocoder/ParallelWaveGAN/egs/V006/voc1/exp/train_nodev_parallel_wavegan.v1/stats.h5
+win_length: null
+window: hann

ppg_sxliu_decoder_V006/stats.h5 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5ff1d3f33879137cf5f6d8d2d9ad6f1db5b1a5ff2a8093e52969976759152bee
+size 4736

pwg.16k_256/checkpoint-400000steps.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b76381a7fdbb799e183d5db949a0eee7fa6f97a3f47412f9ae1eaed05fcd915f
+size 17668782

pwg.16k_256/config.yml ADDED Viewed

	@@ -0,0 +1,104 @@

+allow_cache: true
+batch_max_steps: 25600
+batch_size: 6
+config: conf/parallel_wavegan.v1.yaml
+dev_dumpdir: dump/dev/norm
+dev_feats_scp: null
+dev_segments: null
+dev_wav_scp: null
+discriminator_grad_norm: 1
+discriminator_optimizer_params:
+  eps: 1.0e-06
+  lr: 5.0e-05
+  weight_decay: 0.0
+discriminator_params:
+  bias: true
+  conv_channels: 64
+  in_channels: 1
+  kernel_size: 3
+  layers: 10
+  nonlinear_activation: LeakyReLU
+  nonlinear_activation_params:
+    negative_slope: 0.2
+  out_channels: 1
+  use_weight_norm: true
+discriminator_scheduler_params:
+  gamma: 0.5
+  step_size: 200000
+discriminator_train_start_steps: 100000
+distributed: false
+eval_interval_steps: 1000
+fft_size: 1024
+fmax: 7600
+fmin: 80
+format: hdf5
+generator_grad_norm: 10
+generator_optimizer_params:
+  eps: 1.0e-06
+  lr: 0.0001
+  weight_decay: 0.0
+generator_params:
+  aux_channels: 80
+  aux_context_window: 2
+  dropout: 0.0
+  gate_channels: 128
+  in_channels: 1
+  kernel_size: 3
+  layers: 30
+  out_channels: 1
+  residual_channels: 64
+  skip_channels: 64
+  stacks: 3
+  upsample_net: ConvInUpsampleNetwork
+  upsample_params:
+    upsample_scales:
+    - 4
+    - 4
+    - 4
+    - 4
+  use_weight_norm: true
+generator_scheduler_params:
+  gamma: 0.5
+  step_size: 200000
+global_gain_scale: 1.0
+hop_size: 256
+lambda_adv: 4.0
+log_interval_steps: 100
+num_mels: 80
+num_save_intermediate_results: 4
+num_workers: 2
+outdir: exp/train_nodev_parallel_wavegan.v1
+pin_memory: true
+pretrain: ''
+rank: 0
+remove_short_samples: true
+resume: ''
+sampling_rate: 16000
+save_interval_steps: 5000
+stft_loss_params:
+  fft_sizes:
+  - 1024
+  - 2048
+  - 512
+  hop_sizes:
+  - 120
+  - 240
+  - 50
+  win_lengths:
+  - 600
+  - 1200
+  - 240
+  window: hann_window
+train_dumpdir: dump/train_nodev/norm
+train_feats_scp: null
+train_max_steps: 400000
+train_segments: null
+train_wav_scp: null
+trim_frame_size: 2048
+trim_hop_size: 512
+trim_silence: false
+trim_threshold_in_db: 60
+verbose: 1
+version: 0.6.2a
+win_length: null
+window: hann

pwg.16k_256/stats.h5 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d73ba94507a1a33eb4bd34180cd98871b88b106c489adec76c41d67104284a69
+size 4736

s3prl-vc-ppg_sxliu/checkpoint-50000steps.pkl ADDED Viewed

	@@ -0,0 +1 @@


1	+ ../../../../../../../.cache/huggingface/hub/models--unilight--accent-conversion-2023/blobs/f5fd4b70e8739d1822a1d3491fdf1f4c4d7ae44f8c1a902cba5510079f44ec5e

s3prl-vc-ppg_sxliu/config.yml ADDED Viewed

	@@ -0,0 +1,61 @@

+additional_config: null
+allow_cache: true
+batch_size: 16
+config: conf/taco2_ar.yaml
+dev_scp: data/TXHC_dev/wav.scp
+dev_spemb_scp: null
+distributed: false
+eval_interval_steps: 1000
+fft_size: 1024
+fmax: 7600
+fmin: 80
+global_gain_scale: 1.0
+grad_norm: 1.0
+hop_size: 256
+init_checkpoint: ''
+log_interval_steps: 100
+main_loss_type: L1Loss
+model_params:
+  ar: true
+  encoder_type: taco2
+  hidden_dim: 1024
+  lstmp_dropout_rate: 0.2
+  lstmp_layernorm: false
+  lstmp_layers: 2
+  lstmp_proj_dim: 256
+  prenet_dim: 256
+  prenet_dropout_rate: 0.5
+  prenet_layers: 2
+model_type: Taco2_AR
+num_mels: 80
+num_save_intermediate_results: 4
+num_workers: 2
+optimizer_params:
+  lr: 0.0001
+optimizer_type: AdamW
+outdir: exp/TXHC_ppg_sxliu_taco2_ar
+pin_memory: true
+rank: 0
+resume: ''
+sampling_rate: 16000
+save_interval_steps: 1000
+scheduler: linear_schedule_with_warmup
+scheduler_params:
+  num_warmup_steps: 4000
+train_max_steps: 50000
+train_scp: data/TXHC_train_1032/wav.scp
+train_spemb_scp: null
+trg_stats: exp/TXHC_ppg_sxliu_taco2_ar/stats.h5
+trim_frame_size: 2048
+trim_hop_size: 512
+trim_silence: false
+trim_threshold_in_db: 60
+upstream: ppg_sxliu
+verbose: 1
+version: 0.2.0
+vocoder:
+  checkpoint: /data/group1/z44476r/Experiments/ParallelWaveGAN/egs/l2-arctic/voc1/exp/train_nodev_TXHC_parallel_wavegan.v1/checkpoint-105000steps.pkl
+  config: /data/group1/z44476r/Experiments/ParallelWaveGAN/egs/l2-arctic/voc1/exp/train_nodev_TXHC_parallel_wavegan.v1/config.yml
+  stats: /data/group1/z44476r/Experiments/ParallelWaveGAN/egs/l2-arctic/voc1/exp/train_nodev_TXHC_parallel_wavegan.v1/stats.h5
+win_length: null
+window: hann

s3prl-vc-ppg_sxliu/stats.h5 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2494f218f2758b6c6c4cd7252970647e8e153bbd31bcdf122030092036f6ef7a
+size 4736