Invalid JSON:Expected property name or '}' in JSONat line 2, column 3
| { | |
| // Base config for PreAlignSLM multi-task training (SenseVoice semantic encoder) | |
| // Override with experiment-specific configs for LoRA/finetune/frozen variants | |
| "trainer_type": "ARCaptionTrainer", | |
| "datasets_type": "parquet_list_dataset_multitask_slm", | |
| "synthesizer_type": "SLMSynthesizer", | |
| "data": { | |
| "audio": { | |
| "sample_rate": 16000 | |
| }, | |
| // data.path supports dict format for multi-task: | |
| // {"asr": ["path/to/ASR.list"], "ser": ["path/to/SER.list"], ...} | |
| // or list format (task auto-inferred from filename): | |
| // ["path/to/ASR_LibriSpeech.list", "path/to/SER_IEMOCAP.list", ...] | |
| // | |
| // After running Phase 0.5 caption generation (generate_captions.sh), | |
| // use the captioned list files below. The captioned parquets contain | |
| // an `acoustic_caption` field per segment, enabling speech+text joint | |
| // input for the ArCap acoustic branch (Scenario 1: fast forced-alignment). | |
| // | |
| // Original (no acoustic captions — acoustic branch receives empty text): | |
| // "asr": ["/cpfs/user/lianshi/data/datalist/all/ASR_LibriSpeech.list"], | |
| // "ser": ["/cpfs/user/lianshi/data/datalist/all/SER_IEMOCAP.list", | |
| // "/cpfs/user/lianshi/data/datalist/all/SER_ESD.list"], | |
| // "gr": ["external_data/GR_combined.list"], | |
| // "st": ["external_data/ST_MuST-C.list"] | |
| "path": { | |
| "asr": ["/cpfs/user/bohanli/workspace/Alalm/external_data/captioned/ASR_LibriSpeech_captioned.list"], | |
| "ser": ["/cpfs/user/bohanli/workspace/Alalm/external_data/captioned/SER_combined_captioned.list"], | |
| "gr": ["/cpfs/user/bohanli/workspace/Alalm/external_data/captioned/GR_combined_captioned.list"], | |
| "st": ["/cpfs/user/bohanli/workspace/Alalm/external_data/captioned/ST_MuST-C_captioned.list"] | |
| }, | |
| "splits": "999,1", | |
| "dataset_conf": { | |
| "parse_conf": { | |
| "sample_rate": 16000, | |
| "high_order_resample": false, | |
| "seed": 1234, | |
| "use_audio": true, | |
| "row_group_batch_size": 256 | |
| }, | |
| "filter_conf": { | |
| "max_duration": 30.0, | |
| "min_duration": 0.2 | |
| }, | |
| "pad_multiple": true, | |
| "pad_multiple_conf": { | |
| "multiple_of": 3840 | |
| }, | |
| "shuffle": true, | |
| "shuffle_conf": { | |
| "shuffle_size": 2000 | |
| }, | |
| "sort": true, | |
| "sort_conf": { | |
| "sort_size": 500 | |
| }, | |
| "batch_conf": { | |
| "batch_type": "static", | |
| "batch_size": 4 | |
| } | |
| }, | |
| "num_workers": 8, | |
| "persistent_workers": true, | |
| "pin_memory": true, | |
| "prefetch_factor": 32 | |
| }, | |
| "model": { | |
| "type": "PreAlignSLM", | |
| "semantic_encoder_type": "sensevoice", | |
| "semantic_model_path": "/root/.cache/modelscope/hub/models/iic/SenseVoiceSmall", | |
| "stage0_projector_ckpt": "/cpfs/user/bohanli/workspace/Alalm/egs/arcap_slm/stage0_asr_projector/checkpoints/model-170000.pt", | |
| // Text-only noise params | |
| "drop_prob": 0.05, | |
| "insert_prob": 0.0, | |
| "smooth_low": 0.0, | |
| "smooth_high": 0.1, | |
| // LLM config (override llm_model_path for 0.5B vs 1.5B) | |
| "LLM": { | |
| "llm_model_path": "/cpfs/user/lianshi/pretrained_models/llms/Qwen2.5-1.5B" | |
| }, | |
| // Finetune defaults: frozen LLM (override in exp configs) | |
| "finetune_llm": false, | |
| "use_lora": false, | |
| // Acoustic encoder (CLSPArCap for paralinguistic features) | |
| "acoustic_recog_encoder": { | |
| "type": "arcap", | |
| "clsp_model_path": "yfyeung/CLSP", | |
| "llm_model_path": "/cpfs/user/lianshi/pretrained_models/llms/Qwen2.5-0.5B", | |
| "pretrained_ckpt": "/cpfs/user/bohanli/workspace/Alalm/egs/arcap/arcap_prefix10_reg1.0_textprob0.5/checkpoints/model-900000.pt", | |
| "load_strict": false, | |
| "caption_token": "<caption>", | |
| "prefix_length": 10, | |
| "finetune_llm": false, | |
| "reg_loss_weight": 1.0, | |
| "use_text_prob": 1.0, | |
| "mapper_hidden_dim": 2048, | |
| "mapper_dropout": 0.1 | |
| }, | |
| // Instruction masking separator | |
| "answer_separator": "\nAnswer:" | |
| }, | |
| "train": { | |
| "ddp": true, | |
| "find_unused_parameters": true, | |
| "static_graph": true, | |
| "random_seed": 250929, | |
| "valid_batch_size": 1, | |
| "optimizer_type": "AdamW", | |
| "optimizer_params": { | |
| "lr": 0.0001, | |
| "betas": [0.9, 0.99], | |
| "eps": 1e-06, | |
| "weight_decay": 0.01 | |
| }, | |
| "scheduler_type": "CosineScheduler", | |
| "scheduler_params": { | |
| "lr_batches": 500000, | |
| "min_lr": 1e-05, | |
| "warmup_batches": 5000, | |
| "warmup_start": 0.5 | |
| }, | |
| "gradclip_value": 2, | |
| "epochs": 1000, | |
| "max_steps": 500000, | |
| "save_summary_steps": 100, | |
| "save_checkpoints_steps": 5000, | |
| "valid_interval": 10000, | |
| "keep_checkpoint_max": 10, | |
| "num_visualization": 10, | |
| "dtype": "bf16" | |
| } | |
| } | |