Hanlin1004's picture
Add DSA-Tokenizer inference assets
78335da verified
Raw
History Blame Contribute Delete
4.98 kB
hydra:
run:
dir: ckpts/gan_finetune_${model.name}_${model.mel_spec.mel_spec_type}_${model.tokenizer}/${now:%Y-%m-%d}/${now:%H-%M-%S}
datasets:
name: Emilia_EN_ZH_clean
path: /home/ma-user/work/gwx/daxintan/code_zhl/DSA-Tokenizer/data/Emilia_EN_ZH_clean_unit_4096
batch_size_per_gpu: 10000
batch_size_type: frame
max_samples: 128
num_workers: 10
dataset_type: CustomDatasetPathMelSpeaker
collate_name: collate_fn_mel_speaker_embedding
preprocessed_mel: False
load_multiple_arrow: True
optim:
epochs: 1000
learning_rate: 1.5e-5
num_warmup_updates: 300
grad_accumulation_steps: 1
max_grad_norm: 1.0
bnb_optimizer: False
model:
name: 20260429_DualCodecCADiT_gan_finetune_exp13_band_split_mel_gan_pretrained_attention_larger_speaker_encoder_loss_froze_tokenizer
tokenizer: unit_4096
arch:
dim: 1024
depth: 18
heads: 16
ff_mult: 2
text_dim: 512
should_extend_text: True
conv_layers: 0
checkpoint_activations: False
add_semantic_norm: True
add_acoustic_norm: False
decode_semantic_for_codec: False
convnext_dim: 1024
semantic_hidden_dim: 1024
add_insnorm_before_fsq: False
semantic_acoustic_combination_type: None
hidden_dim: 1024
encoder_num_layers: 8
encoder_in_dim: 100
encoder_type: mel_cnn
using_only_encoder: True
vq_emb_dim: 14
acoustic_downsample_factor: 4
quantizer_type: fsq
vq_num_quantizers: 1
vq_codebook_size: 16384
vq_codebook_dim: 8
add_acoustic_norm_before_merge_to_semantic: False
using_multi_CA_layers: True
CA_layers: 8
num_heads: 8
if_semantic_acoustic_output_linear: False
if_original_mel: True
if_quantize: True
if_learning_kv: True
dropout_acoustic_prob: 0.0
if_use_speaker_encoder: True
speaker_embedding_dim: 256
augment_speaker_loss: False
use_acoustic_adapter_vae: False
span_dropout_prob: 0.0
span_dropout_span_ratio: 0.0
if_ada_norm_for_ca: True
add_pred_speaker_embedding_in_dit: False
add_speaker_embedding_in_dit: False
semantic_acoustic_drop_cond: 0.0
acoustic_drop_cond: 0.0
drop_text_prob_recon: 0.0
drop_text_prob_infill: 0.0
use_double_CADit: False
if_use_ControlDiT: True
semantic_acoustic_combine: False
get_acoustic_code_by_cross_attention: False
if_use_cadit: False
use_flash_attn: True
up_sample_semantic: True
alignment_mode: strict
up_sample_acoustic: False
acoustic_alignment_model: strict
acoustic_k_rope: True
use_acoustic_adapter25hz: False
fsq_levels: [4, 4, 4, 4, 4, 4, 4, 4]
SEANetEncoder_channels: [512, 1024, 1024, 1024]
ratios: [2, 1, 1]
up_sample_after_fsq: True
pass_mask_to_acoustic_encoder: True
froze_tokenizer: True
mel_spec:
target_sample_rate: 24000
n_mel_channels: 100
hop_length: 480
win_length: 1024
n_fft: 1024
center: True
power: 1
mel_spec_type: flow2gan
vocoder:
is_local: True
local_path: /home/ma-user/work/gwx/daxintan/code_zhl/Flow2GAN/output/exp-finetune-step-4-50hz/epoch-20.pt
train:
audio_drop_prob: 1.0
cond_drop_prob: 0.0
speaker_loss_weight: 1.0
mask_type: mix_suffix
pred_x1: False
scaling_loss: False
infill_task_ratio: 0.5
sampler_type: normal
use_guidance_scale_embed: True
init_checkpoint: ckpts/20260409_distill_stage2_DualCodecCADiT_Emilia_clean_encoder_25hz_fsq_recons_recomb_infilling_flow2gan_pass_mask2encoder_froze_tokenizer_flow2gan_unit_4096/model_10000.pt
init_checkpoint_prefer_ema: True
resume_id: null
gan_base_loss_type: fm
gan_wave_mse_weight: 0.0
gan_discriminator_pretrain_steps: 0
gan_generator_warmup_steps: 0
gan_nfe_steps: 4
gan_guidance_scale: 3.0
gan_sway_sampling_coef: null
gan_disc_learning_rate: 1.0e-4
gan_disc_max_grad_norm: 1.0
gan_adv_loss_weight: 0.0
gan_feat_loss_weight: 0.0
gan_flow2gan_model_name: mel_24k_base_50hz
gan_vocoder_n_timesteps: 4
gan_rates: []
gan_periods: [2, 3, 5, 7, 11]
gan_fft_sizes: [2048, 1024, 512]
wave_gan_enabled: False
mel_gan_enabled: True
mel_gan_disc_learning_rate: 2.0e-5
mel_gan_disc_max_grad_norm: 1.0
mel_gan_adv_loss_weight: 0.025
mel_gan_feat_loss_weight: 0.05
mel_gan_scales: [1, 2, 4]
mel_gan_channels: [32, 64, 128, 256]
mel_gan_bands:
- [0, 100]
- [0, 32]
- [24, 72]
- [64, 100]
mel_gan_normalize: True
speaker_mel_loss_weight: 0.5
speaker_mel_real_loss_weight: 0.0
speaker_mel_encoder_checkpoint: ckpts/mel_speaker_encoder_flow2gan_wavlm_ecapa_attention_xlarge200k_from0/model_280000.pt
speaker_mel_encoder_trainable: False
speaker_mel_encoder_use_checkpoint_config: True
speaker_mel_encoder_strict: True
ckpts:
logger: tensorboard
save_per_updates: 1000
last_per_steps: 2000
save_dir: ckpts/${model.name}_${model.mel_spec.mel_spec_type}_${model.tokenizer}
checkpoint_path: ckpts/${model.name}_${model.mel_spec.mel_spec_type}_${model.tokenizer}