pangkaiyu commited on
Commit
cab74fb
·
verified ·
1 Parent(s): fb34eb0

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitignore +184 -0
  2. .gitmodules +9 -0
  3. Dockerfile +29 -0
  4. LEADERBOARD.md +403 -0
  5. README.md +169 -0
  6. README_zh.md +168 -0
  7. config.yaml +7 -0
  8. constraints.txt +2 -0
  9. draw_wer.py +69 -0
  10. ignore-words.txt +2 -0
  11. log_kimilibri.txt +0 -0
  12. log_mini_-5_to_10_all_new_noise.txt +0 -0
  13. log_mini_ea_-5_to_10_all_new_noise.txt +0 -0
  14. log_mini_ea_-5_to_10_new_noise.txt +0 -0
  15. log_mini_origin1.txt +0 -0
  16. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo.jsonl +0 -0
  17. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_default_performance.json +25 -0
  18. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_wer_details.jsonl +0 -0
  19. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank0.log +8 -0
  20. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank1.log +2 -0
  21. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank2.log +2 -0
  22. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank3.log +2 -0
  23. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank4.log +2 -0
  24. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank5.log +2 -0
  25. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank6.log +2 -0
  26. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank7.log +2 -0
  27. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/Step-Audio-2-mini-all-lora3_voices_dev_test_wer_details.jsonl +0 -0
  28. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank0.log +15 -0
  29. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank1.log +2 -0
  30. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank2.log +2 -0
  31. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank3.log +2 -0
  32. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank4.log +2 -0
  33. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank5.log +2 -0
  34. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank6.log +2 -0
  35. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi.jsonl +0 -0
  36. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_default_performance.json +25 -0
  37. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_wer_details.jsonl +0 -0
  38. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank0.log +6 -0
  39. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank1.log +2 -0
  40. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank2.log +2 -0
  41. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank3.log +2 -0
  42. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank4.log +2 -0
  43. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank5.log +2 -0
  44. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank6.log +2 -0
  45. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank7.log +2 -0
  46. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus.jsonl +0 -0
  47. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_default_performance.json +121 -0
  48. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_wer_details.jsonl +0 -0
  49. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank0.log +5 -0
  50. mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank1.log +2 -0
.gitignore ADDED
@@ -0,0 +1,184 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Byte-compiled / optimized / DLL files
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+ data/test.sh
6
+ # C extensions
7
+ *.so
8
+
9
+ # Distribution / packaging
10
+ .Python
11
+ build/
12
+ develop-eggs/
13
+ dist/
14
+ downloads/
15
+ eggs/
16
+ .eggs/
17
+ lib/
18
+ lib64/
19
+ parts/
20
+ sdist/
21
+ var/
22
+ wheels/
23
+ share/python-wheels/
24
+ *.egg-info/
25
+ .installed.cfg
26
+ *.egg
27
+ MANIFEST
28
+
29
+ # PyInstaller
30
+ # Usually these files are written by a python_goose script from a template
31
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
32
+ *.manifest
33
+ *.spec
34
+
35
+ # Installer logs
36
+ pip-log.txt
37
+ pip-delete-this-directory.txt
38
+
39
+ # Unit test / coverage reports
40
+ htmlcov/
41
+ .tox/
42
+ .nox/
43
+ .coverage
44
+ .coverage.*
45
+ .cache
46
+ nosetests.xml
47
+ coverage.xml
48
+ *.cover
49
+ *.py,cover
50
+ .hypothesis/
51
+ .pytest_cache/
52
+ cover/
53
+
54
+ # Translations
55
+ *.mo
56
+ *.pot
57
+
58
+ # Django stuff:
59
+ *.log
60
+ local_settings.py
61
+ db.sqlite3
62
+ db.sqlite3-journal
63
+
64
+ # Flask stuff:
65
+ instance/
66
+ .webassets-cache
67
+
68
+ # Scrapy stuff:
69
+ .scrapy
70
+
71
+ # Sphinx documentation
72
+ docs/_build/
73
+
74
+ # PyBuilder
75
+ .pybuilder/
76
+ target/
77
+
78
+ # Jupyter Notebook
79
+ .ipynb_checkpoints
80
+
81
+ # IPython
82
+ profile_default/
83
+ ipython_config.py
84
+
85
+ # pyenv
86
+ # For a library or package, you might want to ignore these files since the code is
87
+ # intended to run in multiple environments; otherwise, check them in:
88
+ # .python_goose-version
89
+
90
+ # pipenv
91
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
92
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
93
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
94
+ # install all needed dependencies.
95
+ #Pipfile.lock
96
+
97
+ # poetry
98
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
99
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
100
+ # commonly ignored for libraries.
101
+ # https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
102
+ #poetry.lock
103
+
104
+ # pdm
105
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
106
+ #pdm.lock
107
+ # pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
108
+ # in version control.
109
+ # https://pdm.fming.dev/#use-with-ide
110
+ .pdm.toml
111
+
112
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
113
+ __pypackages__/
114
+
115
+ # Celery stuff
116
+ celerybeat-schedule
117
+ celerybeat.pid
118
+
119
+ # SageMath parsed files
120
+ *.sage.py
121
+
122
+ # Environments
123
+ .env
124
+ .env_*
125
+ .venv
126
+ env/
127
+ venv/
128
+ ENV/
129
+ env.bak/
130
+ venv.bak/
131
+
132
+ # Spyder project settings
133
+ .spyderproject
134
+ .spyproject
135
+
136
+ # Rope project settings
137
+ .ropeproject
138
+
139
+ # mkdocs documentation
140
+ /site
141
+
142
+ # mypy
143
+ .mypy_cache/
144
+ .dmypy.json
145
+ dmypy.json
146
+
147
+ # Pyre type checker
148
+ .pyre/
149
+
150
+ # pytype static type analyzer
151
+ .pytype/
152
+
153
+ # Cython debug symbols
154
+ cython_debug/
155
+
156
+ # PyCharm
157
+ # JetBrains specific template is maintained in a separate JetBrains.gitignore that can
158
+ # be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
159
+ # and can be added to the global gitignore or merged into this file. For a more nuclear
160
+ # option (not recommended) you can uncomment the following to ignore the entire idea folder.
161
+ .idea/
162
+ data/downloaded_datasets/*/*/*.wav
163
+ data/downloaded_datasets/*/*.wav
164
+ data/downloaded_datasets/*.wav
165
+ data/downloaded_datasets
166
+ logs
167
+ *.zip
168
+ conf
169
+ .DS_Store
170
+ .ruff_cache
171
+ .log
172
+ *.jsonl
173
+ *.mp3
174
+ *.json
175
+ *.parquet
176
+ *.progress
177
+ # Vscode
178
+ .vscode
179
+ eval_result/
180
+ debug/
181
+ almeval/models/kimia_hf
182
+
183
+ # .pyc
184
+ *.pyc
.gitmodules ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ [submodule "almeval/models/glm4voice"]
2
+ path = almeval/models/glm4voice
3
+ url = https://github.com/THUDM/GLM-4-Voice.git
4
+ [submodule "almeval/models/stepaudio"]
5
+ path = almeval/models/stepaudio
6
+ url = https://github.com/stepfun-ai/Step-Audio.git
7
+ [submodule "almeval/models/kimi_audio"]
8
+ path = almeval/models/kimi_audio
9
+ url = https://github.com/MoonshotAI/Kimi-Audio.git
Dockerfile ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04
2
+
3
+ WORKDIR /app
4
+
5
+ COPY ./requirements.txt /app/
6
+ RUN apt-get update && apt-get install -y \
7
+ python3.10 \
8
+ python3.10-dev \
9
+ curl \
10
+ sox \
11
+ openssh-server \
12
+ ffmpeg \
13
+ libgl1-mesa-glx \
14
+ git \
15
+ ninja-build \
16
+ && rm -rf /var/lib/apt/lists/*
17
+
18
+ # 安装 pip
19
+ RUN curl https://bootstrap.pypa.io/get-pip.py -o get-pip.py \
20
+ && python3.10 get-pip.py \
21
+ && rm get-pip.py
22
+ RUN pip install -r requirements.txt
23
+ RUN pip install git+https://github.com/huggingface/transformers@v4.51.3-Qwen2.5-Omni-preview
24
+ RUN pip install flash-attn --no-build-isolation
25
+
26
+ # alias python3 as python
27
+ RUN ln -s /usr/bin/python3 /usr/bin/python
28
+
29
+ CMD ["/bin/bash"]
LEADERBOARD.md ADDED
@@ -0,0 +1,403 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Leaderboard
2
+
3
+ ## Automatic Speech Recognition (ASR)
4
+
5
+
6
+ <table>
7
+
8
+ <thead>
9
+ <tr>
10
+ <th>Datasets</th>
11
+ <th>Model</th>
12
+ <th>Performance (WER↓)</th>
13
+ </tr>
14
+ </thead>
15
+ <tbody>
16
+ <tr>
17
+ <td rowspan="5"><b>LibriSpeech</b><br>test-clean | test-other</td>
18
+ <td>Qwen2-Audio-base</td>
19
+ <td>1.74 | 4.04</td>
20
+ </tr>
21
+ <tr>
22
+ <td>Baichuan-base</td>
23
+ <td>3.02 | 6.04</td>
24
+ </tr>
25
+ <tr>
26
+ <td>StepAudio-chat</td>
27
+ <td>3.19 | 10.67</td>
28
+ </tr>
29
+ <tr>
30
+ <td>Qwen2.5-Omni</td>
31
+ <td>2.37 | 4.21</td>
32
+ </tr>
33
+ <tr>
34
+ <td>Kimi-Audio</td>
35
+ <td><b>1.28 | 2.42</b></td>
36
+ </tr>
37
+ <tr>
38
+ <td rowspan="5"><b>Fleurs</b><br>zh | en</td>
39
+ <td>Qwen2-Audio-base</td>
40
+ <td>3.63 | 5.20</td>
41
+ </tr>
42
+ <tr>
43
+ <td>Baichuan-base</td>
44
+ <td>4.15 | 8.07</td>
45
+ </tr>
46
+ <tr>
47
+ <td>StepAudio-chat</td>
48
+ <td>4.26 | 8.56</td>
49
+ </tr>
50
+ <tr>
51
+ <td>Qwen2.5-Omni</td>
52
+ <td>2.92 | <b>4.17</b></td>
53
+ </tr>
54
+ <tr>
55
+ <td>Kimi-Audio</td>
56
+ <td><b>2.69</b> | 4.44</td>
57
+ </tr>
58
+ <tr>
59
+ <td rowspan="5"><b>AISHELL-1</b></td>
60
+ <td>Qwen2-Audio-base</td>
61
+ <td>1.52</td>
62
+ </tr>
63
+ <tr>
64
+ <td>Baichuan-base</td>
65
+ <td>1.93</td>
66
+ </tr>
67
+ <tr>
68
+ <td>StepAudio-chat</td>
69
+ <td>2.14</td>
70
+ </tr>
71
+ <tr>
72
+ <td>Qwen2.5-Omni</td>
73
+ <td>1.13</td>
74
+ </tr>
75
+ <tr>
76
+ <td>Kimi-Audio</td>
77
+ <td><b>0.60</b></td>
78
+ </tr>
79
+ <tr>
80
+ <td rowspan="5"><b>AISHELL-2</b> ios</td>
81
+ <td>Qwen2-Audio-base</td>
82
+ <td>3.08</td>
83
+ </tr>
84
+ <tr>
85
+ <td>Baichuan-base</td>
86
+ <td>3.87</td>
87
+ </tr>
88
+ <tr>
89
+ <td>StepAudio-chat</td>
90
+ <td>3.89</td>
91
+ </tr>
92
+ <tr>
93
+ <td>Qwen2.5-Omni</td>
94
+ <td><b>2.56</b></td>
95
+ </tr>
96
+ <tr>
97
+ <td>Kimi-Audio</td>
98
+ <td><b>2.56</b></td>
99
+ </tr>
100
+ <tr>
101
+ <td rowspan="5"><b>WenetSpeech</b><br>test-meeting | test-net</td>
102
+ <td>Qwen2-Audio-base</td>
103
+ <td>8.40 | 7.64</td>
104
+ </tr>
105
+ <tr>
106
+ <td>Baichuan-base</td>
107
+ <td>13.28 | 10.13</td>
108
+ </tr>
109
+ <tr>
110
+ <td>StepAudio-chat</td>
111
+ <td>10.83 | 9.47</td>
112
+ </tr>
113
+ <tr>
114
+ <td>Qwen2.5-Omni</td>
115
+ <td>7.71 | 6.04</td>
116
+ </tr>
117
+ <tr>
118
+ <td>Kimi-Audio</td>
119
+ <td><b>6.28 | 5.37</b></td>
120
+ </tr>
121
+ <tr>
122
+ <td rowspan="5"><b>Kimi-ASR Internal Testset</b><br>subset1 | subset2</td>
123
+ <td>Qwen2-Audio-base</td>
124
+ <td>2.31 | 3.24</td>
125
+ </tr>
126
+ <tr>
127
+ <td>Baichuan-base</td>
128
+ <td>3.41 | 5.60</td>
129
+ </tr>
130
+ <tr>
131
+ <td>StepAudio-chat</td>
132
+ <td>2.82 | 4.74</td>
133
+ </tr>
134
+ <tr>
135
+ <td>Qwen2.5-Omni</td>
136
+ <td>1.53 | 2.68</td>
137
+ </tr>
138
+ <tr>
139
+ <td>Kimi-Audio</td>
140
+ <td><b>1.42 | 2.44</b></td>
141
+ </tr>
142
+ </tbody>
143
+ </table>
144
+
145
+ ## Audio Understanding
146
+
147
+ <table>
148
+ <thead>
149
+ <tr>
150
+ <th>Datasets</th>
151
+ <th>Model</th>
152
+ <th>Performance↑</th>
153
+ </tr>
154
+ </thead>
155
+ <tbody>
156
+ <tr>
157
+ <td rowspan="6"><b>MMAU</b><br>music | sound | speech</td>
158
+ <td>Qwen2-Audio-base</td>
159
+ <td>58.98 | 69.07 | 52.55</td>
160
+ </tr>
161
+ <tr>
162
+ <td>Baichuan-chat</td>
163
+ <td>49.10 | 59.46 | 42.47</td>
164
+ </tr>
165
+ <tr>
166
+ <td>GLM-4-Voice</td>
167
+ <td>38.92 | 43.54 | 32.43</td>
168
+ </tr>
169
+ <tr>
170
+ <td>StepAudio-chat</td>
171
+ <td>49.40 | 53.75 | 47.75</td>
172
+ </tr>
173
+ <tr>
174
+ <td>Qwen2.5-Omni</td>
175
+ <td><b>62.16</b> | 67.57 | 53.92</td>
176
+ </tr>
177
+ <tr>
178
+ <td>Kimi-Audio</td>
179
+ <td>61.68 | <b>73.27</b> | <b>60.66</b></td>
180
+ </tr>
181
+ <tr>
182
+ <td rowspan="5"><b>ClothoAQA</b><br>test | dev</td>
183
+ <td>Qwen2-Audio-base</td>
184
+ <td>71.73 | 72.63</td>
185
+ </tr>
186
+ <tr>
187
+ <td>Baichuan-chat</td>
188
+ <td>48.02 | 48.16</td>
189
+ </tr>
190
+ <tr>
191
+ <td>StepAudio-chat</td>
192
+ <td>45.84 | 44.98</td>
193
+ </tr>
194
+ <tr>
195
+ <td>Qwen2.5-Omni</td>
196
+ <td><b>72.86</b> | 73.12</td>
197
+ </tr>
198
+ <tr>
199
+ <td>Kimi-Audio</td>
200
+ <td>71.24 | <b>73.18</b></td>
201
+ </tr>
202
+ <tr>
203
+ <td rowspan="5"><b>VocalSound</b></td>
204
+ <td>Qwen2-Audio-base</td>
205
+ <td>93.82</td>
206
+ </tr>
207
+ <tr>
208
+ <td>Baichuan-base</td>
209
+ <td>58.17</td>
210
+ </tr>
211
+ <tr>
212
+ <td>StepAudio-chat</td>
213
+ <td>28.58</td>
214
+ </tr>
215
+ <tr>
216
+ <td>Qwen2.5-Omni</td>
217
+ <td>93.73</td>
218
+ </tr>
219
+ <tr>
220
+ <td>Kimi-Audio</td>
221
+ <td><b>94.85</b></td>
222
+ </tr>
223
+ <tr>
224
+ <td rowspan="5"><b>Nonspeech7k</b></td>
225
+ <td>Qwen2-Audio-base</td>
226
+ <td>87.17</td>
227
+ </tr>
228
+ <tr>
229
+ <td>Baichuan-chat</td>
230
+ <td>59.03</td>
231
+ </tr>
232
+ <tr>
233
+ <td>StepAudio-chat</td>
234
+ <td>21.38</td>
235
+ </tr>
236
+ <tr>
237
+ <td>Qwen2.5-Omni</td>
238
+ <td>69.89</td>
239
+ </tr>
240
+ <tr>
241
+ <td>Kimi-Audio</td>
242
+ <td><b>93.93</b></td>
243
+ </tr>
244
+ <tr>
245
+ <td rowspan="5"><b>MELD</b></td>
246
+ <td>Qwen2-Audio-base</td>
247
+ <td>51.23</td>
248
+ </tr>
249
+ <tr>
250
+ <td>Baichuan-chat</td>
251
+ <td>23.59</td>
252
+ </tr>
253
+ <tr>
254
+ <td>StepAudio-chat</td>
255
+ <td>33.54</td>
256
+ </tr>
257
+ <tr>
258
+ <td>Qwen2.5-Omni</td>
259
+ <td>49.83</td>
260
+ </tr>
261
+ <tr>
262
+ <td>Kimi-Audio</td>
263
+ <td><b>59.13</b></td>
264
+ </tr>
265
+ <tr>
266
+ <td rowspan="5"><b>TUT2017</b></td>
267
+ <td>Qwen2-Audio-base</td>
268
+ <td>33.83</td>
269
+ </tr>
270
+ <tr>
271
+ <td>Baichuan-base</td>
272
+ <td>27.9</td>
273
+ </tr>
274
+ <tr>
275
+ <td>StepAudio-chat</td>
276
+ <td>7.41</td>
277
+ </tr>
278
+ <tr>
279
+ <td>Qwen2.5-Omni</td>
280
+ <td>43.27</td>
281
+ </tr>
282
+ <tr>
283
+ <td>Kimi-Audio</td>
284
+ <td><b>65.25</b></td>
285
+ </tr>
286
+ <tr>
287
+ <td rowspan="5"><b>CochlScene</b><br>test | dev</td>
288
+ <td>Qwen2-Audio-base</td>
289
+ <td>52.69 | 50.96</td>
290
+ </tr>
291
+ <tr>
292
+ <td>Baichuan-base</td>
293
+ <td>34.93 | 34.56</td>
294
+ </tr>
295
+ <tr>
296
+ <td>StepAudio-chat</td>
297
+ <td>10.06 | 10.42</td>
298
+ </tr>
299
+ <tr>
300
+ <td>Qwen2.5-Omni</td>
301
+ <td>63.82 | 63.82</td>
302
+ </tr>
303
+ <tr>
304
+ <td>Kimi-Audio</td>
305
+ <td><b>79.84 | 80.99</b></td>
306
+ </tr>
307
+ </tbody>
308
+ </table>
309
+
310
+ ## Audio-to-Text Chat
311
+
312
+ <table>
313
+ <thead>
314
+ <tr>
315
+ <th>Datasets</th>
316
+ <th>Model</th>
317
+ <th>Performance↑</th>
318
+ </tr>
319
+ </thead>
320
+ <tbody>
321
+ <tr>
322
+ <td rowspan="6"><b>OpenAudioBench</b><br>AlpacaEval | Llama Questions |<br>Reasoning QA | TriviaQA | Web Questions</td>
323
+ <td>Qwen2-Audio-chat</td>
324
+ <td>57.19 | 69.67 | 42.77 | 40.30 | 45.20</td>
325
+ </tr>
326
+ <tr>
327
+ <td>Baichuan-chat</td>
328
+ <td>59.65 | 74.33 | 46.73 | 55.40 | 58.70</td>
329
+ </tr>
330
+ <tr>
331
+ <td>GLM-4-Voice</td>
332
+ <td>57.89 | 76.00 | 47.43 | 51.80 | 55.40</td>
333
+ </tr>
334
+ <tr>
335
+ <td>StepAudio-chat</td>
336
+ <td>56.53 | 72.33 | 60.00 | 56.80 | <b>73.00</b></td>
337
+ </tr>
338
+ <tr>
339
+ <td>Qwen2.5-Omni</td>
340
+ <td>72.76 | 75.33 | <b>63.76</b> | 57.06 | 62.80</td>
341
+ </tr>
342
+ <tr>
343
+ <td>Kimi-Audio</td>
344
+ <td><b>75.73</b> | <b>79.33</b> | 58.02 | <b>62.10 </b> | 70.20</td>
345
+ </tr>
346
+ <tr>
347
+ <td rowspan="6"><b>VoiceBench</b><br>AlpacaEval | CommonEval |<br>SD-QA | MMSU</td>
348
+ <td>Qwen2-Audio-chat</td>
349
+ <td>3.69 | 3.40 | 35.35 | 35.43</td>
350
+ </tr>
351
+ <tr>
352
+ <td>Baichuan-chat</td>
353
+ <td>4.00 | 3.39 | 49.64 | 48.80</td>
354
+ </tr>
355
+ <tr>
356
+ <td>GLM-4-Voice</td>
357
+ <td>4.06 | 3.48 | 43.31 | 40.11</td>
358
+ </tr>
359
+ <tr>
360
+ <td>StepAudio-chat</td>
361
+ <td>3.99 | 2.99 | 46.84 | 28.72</td>
362
+ </tr>
363
+ <tr>
364
+ <td>Qwen2.5-Omni</td>
365
+ <td>4.33 | 3.84 | 57.41 | 56.38</td>
366
+ </tr>
367
+ <tr>
368
+ <td>Kimi-Audio</td>
369
+ <td><b>4.46</b> | <b>3.97</b> | <b>63.12</b> | <b>62.17</b></td>
370
+ </tr>
371
+ <tr>
372
+ <td rowspan="6"><b>VoiceBench</b><br>OpenBookQA | IFEval |<br>AdvBench | Avg</td>
373
+ <td>Qwen2-Audio-chat</td>
374
+ <td>49.01 | 22.57 | 98.85 | 54.72</td>
375
+ </tr>
376
+ <tr>
377
+ <td>Baichuan-chat</td>
378
+ <td>63.30 | 41.32 | 86.73 | 62.51</td>
379
+ </tr>
380
+ <tr>
381
+ <td>GLM-4-Voice</td>
382
+ <td>52.97 | 24.91 | 88.08 | 57.17</td>
383
+ </tr>
384
+ <tr>
385
+ <td>StepAudio-chat</td>
386
+ <td>31.87 | 29.19 | 65.77 | 48.86</td>
387
+ </tr>
388
+ <tr>
389
+ <td>Qwen2.5-Omni</td>
390
+ <td>79.12 | 53.88 | 99.62 | 72.83</td>
391
+ </tr>
392
+ <tr>
393
+ <td>Kimi-Audio</td>
394
+ <td><b>83.52</b> | <b>61.10</b> | <b>100.00</b> | <b>76.93</b></td>
395
+ </tr>
396
+ </tbody>
397
+ </table>
398
+
399
+
400
+
401
+ ## Updates
402
+ - 2025-04-25: Initial leaderboard created
403
+
README.md ADDED
@@ -0,0 +1,169 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Kimi-Audio-Evalkit
2
+
3
+ [中文版本](README_zh.md)
4
+
5
+ ## Introduction
6
+
7
+ Kimi-Audio-Evalkit is an evaluation framework designed for audio large language models. Based on Kimi-Audio-Evalkit, you can quickly implement your own models or datasets and conduct fair comparisons with other open-source models.
8
+
9
+ Our work [Kimi-Audio](https://github.com/MoonshotAI/Kimi-Audio-Evalkit) is evaluated using this framework.
10
+
11
+ See [Leaderboard](./LEADERBOARD.md) for current results.
12
+
13
+ ## Getting Started
14
+
15
+ ### Step1: Get the Code
16
+
17
+ ```bash
18
+ git clone https://github.com/MoonshotAI/Kimi-Audio-Evalkit.git
19
+ cd Kimi-Audio-Evalkit
20
+ git submodule update --init --recursive
21
+ ```
22
+
23
+ ### Step2: Prepare Environment
24
+
25
+ You can directly use our pre-built Docker image. If you need to update the environment, you can modify the Dockerfile and rebuild it.
26
+ ```bash
27
+ docker pull moonshotai/almevalkit:v0.4
28
+ ```
29
+ Typically, you need to mount a local directory as the workspace to ensure evaluation results persist after container exit:
30
+ ```bash
31
+ docker run -it -v $(pwd):/app moonshotai/almevalkit:v0.4 bash
32
+ ```
33
+
34
+ ### Step3: Get Datasets
35
+
36
+ Most datasets used by ALMEvalKit can be downloaded using our included tools. Some datasets cannot be fully automated. Please refer to [Download Datasets](./data/README.md) for details.
37
+ For datasets on Hugging Face, we will soon provide a more direct usage method. Please stay tuned for updates.
38
+
39
+ ### Step4: Configure config.yaml
40
+ You may need to fill in several fields in the config.yaml in the root directory to help us locate your data source. By default, datasets will be downloaded to the data/ directory under the current directory. If you downloaded them elsewhere, please enter the root directory in the dataset_root field.
41
+ ```yaml
42
+ DATASETS:
43
+ dataset_root: "/path/to/your/dataset/root"
44
+ ```
45
+ ### Step5: Evaluation
46
+
47
+ run_audio.sh is the entry point for evaluation. You can get help using `--help`
48
+
49
+ For example, to evaluate Kimi-Audio on all datasets:
50
+ ```
51
+ bash run_audio.sh --model Kimi-Audio --data all --skip-eval
52
+ ```
53
+ By default, inference results, evaluation results, and metric reports will be generated in the eval_results directory under the current directory. You can change this behavior by passing --work-dir.
54
+
55
+ Using --skip-eval allows the model to only perform inference without evaluation, which helps keep your GPU running efficiently.
56
+ After inference is complete, you can run the command again to start evaluation. You can add the --reeval parameter to force re-evaluation of the dataset, which won't trigger re-inference but will regenerate the metric report.
57
+
58
+ Note: Our default LLM method is gpt-4o-mini. You need to set your own API KEY to enable it. We will support more evaluation models in the future.
59
+ ```
60
+ export OPENAI_API_KEY=your_api_key
61
+ bash run_audio.sh --model Kimi-Audio --data all --reeval
62
+ ```
63
+
64
+ Currently supported models, datasets, and evaluation models are listed below:
65
+
66
+ **Models**
67
+
68
+ - **Baichuan Series**: Baichuan-Audio-Base, Baichuan-Audio-Instruct
69
+ - **Qwen Series**: Qwen2-Audio-7B, Qwen2-Audio-7B-Instruct, Qwen2.5-Omni-7B
70
+ - **GLM Series**: GLM4-Voice
71
+ - **Others**: StepAudio, Kimi-Audio
72
+
73
+ **Datasets**
74
+
75
+ | Dataset Category | Datasets |
76
+ |-----------------|----------|
77
+ | ASR | LibriSpeech, Fleurs-zh, Fleurs-en, AISHELL-1, AISHELL-2, WenetSpeech |
78
+ | MQA | mmau-test-mini, openbookqa, mmsu, MELD, Nonspeech7k, TUT2017, VocalSound, CochlScene |
79
+ | OpenQA | alpacaeval_full, commoneval, advbench, ifeval |
80
+ | RefQA | ClothoAQA, sd-qa, OpenAudioBench |
81
+
82
+ - For more information about dataset types, ownership, etc., please check the implementation of the relevant datasets.
83
+
84
+ ## Adding Datasets
85
+
86
+ We believe the greatest value of ALMEvalKit is not in reproducing existing results, but in providing a simple mechanism to help users add their own datasets and models, and conduct fair comparisons with other model results.
87
+
88
+ We strongly recommend first reading [Dataset Definition](./almeval/datasets/base.py) to understand how we classify datasets. This will help you correctly set the meta information for new datasets, ensuring they are used appropriately.
89
+
90
+ To add a dataset, you need to write a few lines of code to prepare a jsonl file named dataset_name.jsonl for ALMEvalKit. Each line of the jsonl is a json record, and we require each line to have the following fields:
91
+ ```
92
+ {
93
+ "index": int, # unique identifier for a piece of data
94
+ "audio_path": str | list[str], # audio location
95
+ "question": str, # question or instruction for the audio, e.g., "Please transcribe the audio content into text". Set to empty if not needed
96
+ "answer": str, # ground truth answer. Set to empty if not needed (e.g., for Open-QA)
97
+ "subset": str, # subset name. Sometimes a dataset can be split into several subsets, which will be evaluated separately and reported independently. If you don't have subsets, use the dataset name
98
+ }
99
+
100
+ For Audio-QA type datasets, we require an additional "audio_content" field to provide the content in text form for LLM evaluation models to assess answer correctness:
101
+ {
102
+ "index": int, # unique identifier for a piece of data
103
+ "audio_path": str | list[str], # audio location
104
+ "question": str, # question or instruction for the audio
105
+ "audio_content": str, # text form of the audio
106
+ "answer": str, # ground truth answer
107
+ "subset": str, # subset name
108
+ }
109
+ ```
110
+ [Download Dataset](./data/download_benchmark.py) shows how we download & process data, which you can refer to.
111
+
112
+ After completing this file, you can add your dataset to the appropriate category. Generally, by inheriting the parent class of that category and filling in some fields, your dataset will be ready to use. For example:
113
+ ```
114
+ class Vocalsound(AudioMQADataset):
115
+ DATASET_NAME = 'VocalSound'
116
+ DATASET_SERIES = 'VocalSound'
117
+ AUDIO_TYPE = 'AudioEvent'
118
+ ```
119
+ This indicates that the vocalsound dataset is an MQA dataset (multiple-choice questions), it belongs to the vocalsound dataset series, its AUDIO_TYPE is marked as "AudioEvent", indicating that this dataset is related to sound events (non-speech), which will affect some model evaluation behaviors during evaluation.
120
+
121
+ If you download it to the dataset cache directory, you can now evaluate this dataset on any model:
122
+ ```bash
123
+ bash run_audio.sh --model Kimi-Audio --data vocalsound
124
+ ```
125
+ If you saved this file elsewhere, please tell us in config.yaml:
126
+ ```yaml
127
+ DATASETS:
128
+ dataset_root: "/path/to/your/dataset/root"
129
+ datasets:
130
+ #example:
131
+ example: "/path/to/your/dataset/example.jsonl"
132
+ Vocalsound: "/path/to/your/dataset/VocalSound.jsonl"
133
+ ```
134
+
135
+ ## Adding Models
136
+
137
+ Evaluating your model in ALMEvalKit is also very easy. You only need to implement the generate_inner method, whose signature is:
138
+
139
+ ```
140
+ def generate_inner(self, msg:dict) -> (str, str)
141
+ ```
142
+ **msg** is a piece of data from the dataset, with the following format:
143
+ ```python
144
+ {
145
+ "index": int, # the index of a piece of data in the dataset, as above
146
+ "audio": list[str] # in most cases, the length is 1, and audio[0] can be used to get the audio for this piece of data
147
+ "text": str # the "question" field of the data, may be empty
148
+ "meta": dict # dataset meta information, such as audio_type, name, task, etc. If a piece of data has a meta field, it will also be included here
149
+ }
150
+ ```
151
+
152
+ This function returns prompt:str, result:str, where prompt is the actual text sent to the model for inference, and result is the model's inference result.
153
+
154
+ **Note** "The actual text sent to the model for inference" is not necessarily equal to `msg['text']`, as we can set rules at runtime to modify it. Usually, we implement a `get_prompt(msg) -> text` to handle this.
155
+
156
+ **The best way to add a model is to copy an already implemented model and follow its pattern**
157
+
158
+ ## Call for Contribution
159
+
160
+ We hope the community can work together to build a fair, efficient, and unified audio large language model evaluation framework in the following aspects:
161
+
162
+ - Add features, fix bugs, improve code quality and usability
163
+ - Support more models and datasets
164
+ - Improve readability, contribute examples and docs
165
+
166
+ Due to limited information, we cannot find the best prompts for each model across different tasks/datasets. We also welcome the community to provide best practices, making our leaderboard better reflect the model's true capabilities.
167
+
168
+ We also recommend that you use pre-commit hooks to auto-format your code. see [pre-commit](https://pre-commit.com/)
169
+
README_zh.md ADDED
@@ -0,0 +1,168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Kimi-Audio-Evalkit
2
+
3
+ [English Version](README.md)
4
+
5
+ ## 介绍
6
+
7
+ Kimi-Audio-Evalkit是一个为音频大模型评测设计的评测框架,基于Kimi-Audio-Evalkit,你可以快速实现自己的模型或数据集,并公平的与其他开源模型进行比对。
8
+
9
+ 我们的工作[Kimi-Audio](https://github.com/MoonshotAI/Kimi-Audio-Evalkit)基于此框架评测。
10
+
11
+ [Leaderboard](./LEADERBOARD.md)是目前的评测结果。
12
+
13
+ ## 开始评测
14
+
15
+ ### Step1: 获取代码
16
+
17
+ ```bash
18
+ git clone https://github.com/MoonshotAI/Kimi-Audio-Evalkit.git
19
+ cd Kimi-Audio-Evalkit
20
+ git submodule update --init --recursive
21
+ ```
22
+
23
+ ### Step2: 准备环境
24
+
25
+ 你可以直接使用我们预先build好的镜像,如果你需要更新镜像环境,可以修改Dockerfile后重新构造
26
+ ```bash
27
+ docker pull moonshotai/almevalkit:v0.4
28
+ ```
29
+ 通常情况你需要mount本地目录,并将其作为工作目录,以便评测结果在容器退出后仍然存在
30
+ ```bash
31
+ docker run -it -v $(pwd):/app moonshotai/almevalkit:v0.4 bash
32
+ ```
33
+
34
+ ### Step3: 获取数据集
35
+
36
+ ALMEvalKit所用的大部分数据集都可以通过我们附带的工具下载,有些数据集不能全自动执行,具体请参阅[下载数据集](./data/README.md)
37
+ 对位于huggingface的数据集,我们很快将提供更直接的使用方式,请关注更新。
38
+
39
+ ### Step4: 配置config.yaml
40
+ 你也许需要填写根目录下的config.yaml中的若干字段,帮助我们找到你的数据源。默认情况下,数据集会被下载到当前目录的data/downloaded_datasets下,如果你下载到了其他地方,请将根目录填入dataset_root字段。
41
+ ```yaml
42
+ DATASETS:
43
+ dataset_root: "/path/to/your/dataset/root"
44
+ ```
45
+
46
+ ### Step5: 评测
47
+
48
+ run_audio.sh为评测入口,你可以通过`--help`取得帮助
49
+
50
+ 例如,我们希望跑Kimi-Audio在全部数据集上的结果:
51
+ ```
52
+ bash run_audio.sh --model Kimi-Audio --data all --skip-eval
53
+ ```
54
+ 默认情况下,推理结果文件、评测结果文件、指标报告文件将生成在当前目录的eval_results目录下,你可以通过传递--work-dir改变这一行为。
55
+
56
+ 使用--skip-eval可以让模型只推理,不评测,这样有助于保持你的GPU高效运转。
57
+ 推理完毕后,你只需要重新运行一次,即可展开评测,你可以通过添加--reeval参数来强制对数据集重新评测,这不会触发重新推理,但会重新生成指标报告。
58
+
59
+ Note: 我们默认的LLM方式是gpt-4o-mini,你需要设定你自己的API KEY来启用。未来我们将支持更多评测模型。
60
+ ```
61
+ export OPENAI_API_KEY=your_api_key
62
+ bash run_audio.sh --model Kimi-Audio --data all --reeval
63
+ ```
64
+
65
+ 目前已经支持的模型、数据集和评测模型列表如下
66
+
67
+ **模型**
68
+
69
+ - **Baichuan Series**: Baichuan-Audio-Base, Baichuan-Audio-Instruct
70
+ - **Qwen Series**: Qwen2-Audio-7B, Qwen2-Audio-7B-Instruct, Qwen2.5-Omni-7B
71
+ - **GLM Series**: GLM4-Voice
72
+ - **Others**: StepAudio, Kimi-Audio
73
+
74
+ **数据集**
75
+ | 数据集类型 | 数据集 |
76
+ |-----------------|----------|
77
+ | ASR | LibriSpeech, Fleurs-zh, Fleurs-en, AISHELL-1, AISHELL-2, WenetSpeech |
78
+ | MQA | mmau-test-mini, openbookqa, mmsu, MELD, Nonspeech7k, TUT2017, VocalSound, CochlScene |
79
+ | OpenQA | alpacaeval_full, commoneval, advbench, ifeval |
80
+ | RefQA | ClothoAQA, sd-qa, OpenAudioBench |
81
+
82
+ - 数据集的类型、归属等更多信息,可以查看相关数据集的实现。
83
+
84
+ ## 添加数据集
85
+
86
+ 我们相信ALMEvalKit的最大价值不是复现某个已有结果,而是提供一种简单的机制帮助用户添加自己的数据集和模型,并能够与其他模型结果公平比较。
87
+
88
+ 我们强烈建议首先阅读[数据集的定义](./almeval/datasets/base.py)了解我们如何对数据集分类,这将帮助你正确的设定新数据集的meta信息,使它们被更正确的使用。
89
+
90
+ 要添加数据集,你需要写几行代码,为ALMEvalKit准备一个名为dataset_name.jsonl的jsonl文件。jsonl的每一行是一个json记录,我们要求每一行必须具有的字段是:
91
+ ```
92
+ {
93
+ "index": int, # 一条数据的唯一标识
94
+ "audio_path": str | list[str], # 音频位置
95
+ "question": str, # 针对音频的问题或指令,例如"请将音频内容转写为文字",如果你不需要此字段,请设为空
96
+ "answer": str, # ground truth答案,如果你不需要此字段(如Open-QA),请设为空
97
+ "subset": str, # 子数据,有时候一个数据集可以被切分为若干个子集,这些子集将被分别评估,独立汇报结果。如果你没有子数据集,填数据集名字即可
98
+ }
99
+
100
+ 对于Audio-QA类的数据集,我们要求额外增加一个"audio_content"字段,以文字形式写出内容,以便交给LLM评测模型评测答案是否正确。
101
+ {
102
+ "index": int, # 一条数据的唯一标识
103
+ "audio_path": str | list[str], # 音频位置
104
+ "question": str, # 针对音频的问题或指令,例如"请将音频内容转写为文字",如果你不需要此字段,请设为空
105
+ "audio_content": str, # 音频��文本形式
106
+ "answer": str, # ground truth答案,如果你不需要此字段(如Open-QA),请设为空
107
+ "subset": str, # 子数据,有时候一个数据集可以被切分为若干个子集,这些子集将被分别评估,独立汇报结果。如果你没有子数据集,填数据集名字即可
108
+ }
109
+ ```
110
+ [下载数据集](./data/download_benchmark.py)表明了我们如何下载&处理数据,你可以拿来参考。
111
+
112
+ 完成此文件后,你可以将你的数据集添加到适当的类别下,一般而言,继承此类别的父类并填写一些字段后,你的数据集就可用了。例如:
113
+ ```
114
+ class Vocalsound(AudioMQADataset):
115
+ DATASET_NAME = 'VocalSound'
116
+ DATASET_SERIES = 'VocalSound'
117
+ AUDIO_TYPE = 'AudioEvent'
118
+ ```
119
+ 这表明,数据集vocalsound是一个MQA数据集(单选题),它所属的数据集系列是vocalsound,它的AUDIO_TYPE标记为"AudioEvent",说明此数据集是与声音事件(非语音)有关的数据集,这将会在评测时影响一些模型的评测行为。
120
+
121
+ 如果你将其下载到数据集缓存目录下,现在你就可以在任意模型上评测此数据集了
122
+ ```bash
123
+ bash run_audio.sh --model Kimi-Audio --data vocalsound
124
+ ```
125
+ 如果你将此文件保存在了其他位置,请在config.yaml中告诉我们
126
+ ```yaml
127
+ DATASETS:
128
+ dataset_root: "/path/to/your/dataset/root"
129
+ datasets:
130
+ #example:
131
+ example: "/path/to/your/dataset/example.jsonl"
132
+ Vocalsound: "/path/to/your/dataset/VocalSound.jsonl"
133
+ ```
134
+
135
+ ## 添加模型
136
+
137
+ 在ALMEvalKit评测你的模型也十分容易,你只需要实现generate_inner方法即可,此方法的签名是:
138
+
139
+ ```
140
+ def generate_inner(self, msg:dict) -> (str, str)
141
+ ```
142
+ **msg** 就是从数据集中的一条数据,它的格式是:
143
+ ```python
144
+ {
145
+ "index": int, # 即数据集中一条数据的index,见上
146
+ "audio": list[str] # 大部分情况下长度是1,取audio[0]即可获得此条数据的音频
147
+ "text": str # 即数据的"question"字段,可能为空
148
+ "meta": dict # 数据集的meta信息,如audio_type, name,task等存在这里,如果数据集的一条数据有meta字段,也将会被吸入此字段中
149
+ }
150
+ ```
151
+
152
+ 此函数的返回是 prompt:str, result:str,prompt为实际送入模型推理的文本,result为模型推理结果。
153
+
154
+ **注意** "实际送入模型推理的文本"不一定等于`msg['text']`,因为我们可以在运行时设定规则篡改它,通常我们会实现一个`get_prompt(msg) -> text`来做这件事。
155
+
156
+ **添加一个模型的最好方式就是copy一个已经实现的模型照猫画虎**
157
+
158
+ ## Call for contribution
159
+
160
+ 我们希望社区在如下方面共建一个公平、高效、统一的音频大模型评测框架
161
+
162
+ - 增加功能,修改bug,提高代码质量和易用性
163
+ - 支持更多模型和数据集
164
+ - 提高可读性,贡献examples和docs
165
+
166
+ 受限于我们所掌握的信息,我们无法为每个模型找到不同任务/数据集下的最佳prompt,我们也欢迎社区提供最佳实践,使得我们的leaderboard能够更加真实的反应模型的极限能力。
167
+
168
+ 我们推荐使用[pre-commit](https://pre-commit.com/)来自动格式化你的代码,使你的代码规范与项目保持一致。
config.yaml ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ DATASETS:
2
+ dataset_root: "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets"
3
+ datasets:
4
+ #example:
5
+ example: "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets/LibriSpeech/LibriSpeech.jsonl"
6
+
7
+
constraints.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ transformers==4.49.0
2
+ tokenizers==0.21.4
draw_wer.py ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import json
2
+ from json import JSONDecodeError
3
+ import matplotlib.pyplot as plt
4
+
5
+ jsonl_path = "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/lora_-5_to_10__7B/Qwen2.5-Omni-7B/kimi_10k_noise_-5_to_10_linear_val5_abs/Qwen2.5-Omni-7B_kimi_10k_noise_-5_to_10_linear_val5_abs_wer_details.jsonl" # 改成你的文件
6
+ out_png = "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/pic/-5_to_10_val5.png" # 保存路径
7
+
8
+ dec = json.JSONDecoder()
9
+
10
+ def iter_json_from_line(line: str):
11
+ line = line.strip()
12
+ if not line:
13
+ return
14
+ try:
15
+ yield json.loads(line)
16
+ return
17
+ except JSONDecodeError as e:
18
+ # 典型:一行里粘了多个 JSON(Extra data)
19
+ i, n = 0, len(line)
20
+ while i < n:
21
+ while i < n and line[i].isspace():
22
+ i += 1
23
+ if i >= n:
24
+ break
25
+ obj, j = dec.raw_decode(line, i)
26
+ yield obj
27
+ i = j
28
+
29
+ xs, ys = [], []
30
+ cnt_obj = 0
31
+
32
+ with open(jsonl_path, "r", encoding="utf-8") as f:
33
+ for ln, line in enumerate(f, 1):
34
+ try:
35
+ for obj in iter_json_from_line(line):
36
+ cnt_obj += 1
37
+ x = obj.get("index", cnt_obj)
38
+
39
+ y = obj.get("utt_wer", None)
40
+ if y is None:
41
+ y = (obj.get("wer_details") or {}).get("utt_wer", None)
42
+ if y is None:
43
+ continue
44
+
45
+ y = float(y)
46
+ # 兼容:0~1 or 0~100
47
+ if y <= 1.0:
48
+ y *= 100.0
49
+
50
+ xs.append(x)
51
+ ys.append(y)
52
+ except Exception as e:
53
+ # 定位到底是哪一行坏了
54
+ print(f"[ERROR] line {ln} parse failed: {repr(e)}")
55
+ print("line snippet:", repr(line[:200]))
56
+ raise
57
+
58
+ print(f"parsed {cnt_obj} json objects, plotted {len(xs)} points")
59
+
60
+ plt.figure()
61
+ plt.scatter(xs, ys, s=8)
62
+ plt.xlabel("index")
63
+ plt.ylabel("utt_wer (%)")
64
+ plt.ylim(0, 120)
65
+ plt.title("Per-utterance WER scatter")
66
+ plt.savefig(out_png, dpi=300, bbox_inches="tight")
67
+ plt.close()
68
+
69
+ print("saved to:", out_png)
ignore-words.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ rouge
2
+ Rouge
log_kimilibri.txt ADDED
The diff for this file is too large to render. See raw diff
 
log_mini_-5_to_10_all_new_noise.txt ADDED
The diff for this file is too large to render. See raw diff
 
log_mini_ea_-5_to_10_all_new_noise.txt ADDED
The diff for this file is too large to render. See raw diff
 
log_mini_ea_-5_to_10_new_noise.txt ADDED
The diff for this file is too large to render. See raw diff
 
log_mini_origin1.txt ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_default_performance.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "task": "ASR",
3
+ "dataset": "voices_dev_clo",
4
+ "model": "Step-Audio-2-mini-all-lora3",
5
+ "date": "2026-01-04 15:40:25.671810",
6
+ "performance": {
7
+ "babb": {
8
+ "wer": 1.76,
9
+ "total": 364
10
+ },
11
+ "musi": {
12
+ "wer": 1.51,
13
+ "total": 371
14
+ },
15
+ "none": {
16
+ "wer": 1.48,
17
+ "total": 380
18
+ },
19
+ "tele": {
20
+ "wer": 1.43,
21
+ "total": 351
22
+ }
23
+ },
24
+ "eval_method": "qwen2-audio-impl"
25
+ }
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_wer_details.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank0.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ 2026-01-04 12:52:12 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:52:12 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0112/Lab41-SRI-VOiCES-rm1-babb-sp0112-ch123215-sg0025-mc01-stu-clo-dg080.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
3
+ 2026-01-04 12:58:29 | INFO | waiting for other ranks to finish, time elapsed: 10s
4
+ 2026-01-04 12:58:39 | INFO | waiting for other ranks to finish, time elapsed: 20s
5
+ 2026-01-04 12:58:49 | INFO | waiting for other ranks to finish, time elapsed: 30s
6
+ 2026-01-04 12:58:59 | INFO | waiting for other ranks to finish, time elapsed: 40s
7
+ 2026-01-04 12:58:59 | INFO | model Step-Audio-2-mini-all-lora3, data voices_dev_clo, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo.jsonl.
8
+ 2026-01-04 12:58:59 | INFO | skip eval for voices_dev_clo
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank1.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:50:25 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:50:25 | INFO | Msg example: {'index': 2, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0122/Lab41-SRI-VOiCES-rm1-babb-sp0122-ch121729-sg0002-mc02-lav-clo-dg060.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank2.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:52:11 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:52:11 | INFO | Msg example: {'index': 3, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0122/Lab41-SRI-VOiCES-rm1-babb-sp0122-ch121730-sg0014-mc01-stu-clo-dg000.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank3.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:51:24 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:51:24 | INFO | Msg example: {'index': 4, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0159/Lab41-SRI-VOiCES-rm1-babb-sp0159-ch135897-sg0052-mc01-stu-clo-dg100.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank4.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:51:06 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:51:06 | INFO | Msg example: {'index': 5, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0174/Lab41-SRI-VOiCES-rm1-babb-sp0174-ch084280-sg0013-mc02-lav-clo-dg010.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank5.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:51:33 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:51:33 | INFO | Msg example: {'index': 6, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0188/Lab41-SRI-VOiCES-rm1-babb-sp0188-ch135249-sg0029-mc01-stu-clo-dg170.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank6.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:50:37 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:50:37 | INFO | Msg example: {'index': 7, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0205/Lab41-SRI-VOiCES-rm1-babb-sp0205-ch159056-sg0032-mc01-stu-clo-dg020.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank7.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:50:40 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
2
+ 2026-01-04 12:50:40 | INFO | Msg example: {'index': 8, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0208/Lab41-SRI-VOiCES-rm1-babb-sp0208-ch126851-sg0011-mc02-lav-clo-dg070.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/Step-Audio-2-mini-all-lora3_voices_dev_test_wer_details.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank0.log ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-01-04 12:24:04 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:24:04 | INFO | Msg example: {'index': 0, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032639-sg0029-mc01-stu-clo-dg160.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
3
+ 2026-01-04 12:50:32 | INFO | waiting for other ranks to finish, time elapsed: 10s
4
+ 2026-01-04 12:50:42 | INFO | waiting for other ranks to finish, time elapsed: 20s
5
+ 2026-01-04 12:50:52 | INFO | waiting for other ranks to finish, time elapsed: 30s
6
+ 2026-01-04 12:51:02 | INFO | waiting for other ranks to finish, time elapsed: 40s
7
+ 2026-01-04 12:51:12 | INFO | waiting for other ranks to finish, time elapsed: 50s
8
+ 2026-01-04 12:51:22 | INFO | waiting for other ranks to finish, time elapsed: 60s
9
+ 2026-01-04 12:51:32 | INFO | waiting for other ranks to finish, time elapsed: 70s
10
+ 2026-01-04 12:51:42 | INFO | waiting for other ranks to finish, time elapsed: 80s
11
+ 2026-01-04 12:51:52 | INFO | waiting for other ranks to finish, time elapsed: 90s
12
+ 2026-01-04 12:52:02 | INFO | waiting for other ranks to finish, time elapsed: 100s
13
+ 2026-01-04 12:52:12 | INFO | waiting for other ranks to finish, time elapsed: 110s
14
+ 2026-01-04 12:52:12 | INFO | model Step-Audio-2-mini-all-lora3, data voices_dev_test, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/Step-Audio-2-mini-all-lora3_voices_dev_test.jsonl.
15
+ 2026-01-04 12:52:12 | INFO | skip eval for voices_dev_test
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank1.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:23:40 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:23:40 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032639-sg0029-mc05-stu-far-dg160.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-far'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank2.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:24:00 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:24:00 | INFO | Msg example: {'index': 2, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032658-sg0012-mc05-stu-far-dg070.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-far'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank3.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:23:47 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:23:47 | INFO | Msg example: {'index': 3, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032658-sg0012-mc01-stu-clo-dg070.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank4.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:23:49 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:23:49 | INFO | Msg example: {'index': 4, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp1447/Lab41-SRI-VOiCES-rm4-babb-sp1447-ch130550-sg0026-mc01-stu-clo-dg010.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank5.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:23:46 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:23:46 | INFO | Msg example: {'index': 5, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp1447/Lab41-SRI-VOiCES-rm4-babb-sp1447-ch130550-sg0026-mc05-stu-far-dg010.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-far'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank6.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 12:23:39 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
2
+ 2026-01-04 12:23:39 | INFO | Msg example: {'index': 6, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp1447/Lab41-SRI-VOiCES-rm4-babb-sp1447-ch130551-sg0027-mc01-stu-clo-dg140.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_default_performance.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "task": "ASR",
3
+ "dataset": "chime4_dev-real_kimi",
4
+ "model": "Step-Audio-2-mini-all-lora4",
5
+ "date": "2026-01-04 15:40:41.147211",
6
+ "performance": {
7
+ "bus": {
8
+ "wer": 4.41,
9
+ "total": 410
10
+ },
11
+ "caf": {
12
+ "wer": 4.05,
13
+ "total": 410
14
+ },
15
+ "ped": {
16
+ "wer": 3.76,
17
+ "total": 410
18
+ },
19
+ "str": {
20
+ "wer": 4.0,
21
+ "total": 410
22
+ }
23
+ },
24
+ "eval_method": "qwen2-audio-impl"
25
+ }
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_wer_details.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank0.log ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ 2026-01-04 13:01:58 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:58 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C0103_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
3
+ 2026-01-04 13:04:56 | INFO | waiting for other ranks to finish, time elapsed: 10s
4
+ 2026-01-04 13:05:06 | INFO | waiting for other ranks to finish, time elapsed: 20s
5
+ 2026-01-04 13:05:06 | INFO | model Step-Audio-2-mini-all-lora4, data chime4_dev-real_kimi, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi.jsonl.
6
+ 2026-01-04 13:05:06 | INFO | skip eval for chime4_dev-real_kimi
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank1.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:47 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:47 | INFO | Msg example: {'index': 2, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C0105_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank2.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:51 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:51 | INFO | Msg example: {'index': 3, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010C_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank3.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:49 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:49 | INFO | Msg example: {'index': 4, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010G_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank4.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:49 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:49 | INFO | Msg example: {'index': 5, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010J_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank5.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:48 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:48 | INFO | Msg example: {'index': 6, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010K_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank6.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:50 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:50 | INFO | Msg example: {'index': 7, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010L_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank7.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:01:51 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
2
+ 2026-01-04 13:01:51 | INFO | Msg example: {'index': 8, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010O_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_default_performance.json ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "task": "ASR",
3
+ "dataset": "noizeus",
4
+ "model": "Step-Audio-2-mini-all-lora4",
5
+ "date": "2026-01-04 15:40:40.185443",
6
+ "performance": {
7
+ "airport_0dB": {
8
+ "wer": 25.21,
9
+ "total": 30
10
+ },
11
+ "airport_10dB": {
12
+ "wer": 2.07,
13
+ "total": 30
14
+ },
15
+ "airport_15dB": {
16
+ "wer": 1.65,
17
+ "total": 30
18
+ },
19
+ "airport_5dB": {
20
+ "wer": 9.92,
21
+ "total": 30
22
+ },
23
+ "babble_0dB": {
24
+ "wer": 38.43,
25
+ "total": 30
26
+ },
27
+ "babble_10dB": {
28
+ "wer": 3.31,
29
+ "total": 30
30
+ },
31
+ "babble_15dB": {
32
+ "wer": 2.07,
33
+ "total": 30
34
+ },
35
+ "babble_5dB": {
36
+ "wer": 4.96,
37
+ "total": 30
38
+ },
39
+ "car_0dB": {
40
+ "wer": 35.54,
41
+ "total": 30
42
+ },
43
+ "car_10dB": {
44
+ "wer": 5.79,
45
+ "total": 30
46
+ },
47
+ "car_15dB": {
48
+ "wer": 2.48,
49
+ "total": 30
50
+ },
51
+ "car_5dB": {
52
+ "wer": 7.85,
53
+ "total": 30
54
+ },
55
+ "exhibition_0dB": {
56
+ "wer": 31.4,
57
+ "total": 30
58
+ },
59
+ "exhibition_10dB": {
60
+ "wer": 4.55,
61
+ "total": 30
62
+ },
63
+ "exhibition_15dB": {
64
+ "wer": 2.48,
65
+ "total": 30
66
+ },
67
+ "exhibition_5dB": {
68
+ "wer": 7.44,
69
+ "total": 30
70
+ },
71
+ "restaurant_0dB": {
72
+ "wer": 30.17,
73
+ "total": 30
74
+ },
75
+ "restaurant_10dB": {
76
+ "wer": 1.65,
77
+ "total": 30
78
+ },
79
+ "restaurant_15dB": {
80
+ "wer": 2.89,
81
+ "total": 30
82
+ },
83
+ "restaurant_5dB": {
84
+ "wer": 6.61,
85
+ "total": 30
86
+ },
87
+ "station_0dB": {
88
+ "wer": 23.97,
89
+ "total": 30
90
+ },
91
+ "station_10dB": {
92
+ "wer": 3.31,
93
+ "total": 30
94
+ },
95
+ "station_15dB": {
96
+ "wer": 2.07,
97
+ "total": 30
98
+ },
99
+ "station_5dB": {
100
+ "wer": 9.92,
101
+ "total": 30
102
+ },
103
+ "street_0dB": {
104
+ "wer": 39.26,
105
+ "total": 30
106
+ },
107
+ "street_10dB": {
108
+ "wer": 5.37,
109
+ "total": 30
110
+ },
111
+ "street_15dB": {
112
+ "wer": 1.65,
113
+ "total": 30
114
+ },
115
+ "street_5dB": {
116
+ "wer": 13.64,
117
+ "total": 30
118
+ }
119
+ },
120
+ "eval_method": "qwen2-audio-impl"
121
+ }
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_wer_details.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank0.log ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ 2026-01-04 13:00:54 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: noizeus
2
+ 2026-01-04 13:00:54 | INFO | Msg example: {'index': 0, 'audio': ['/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets/noizeus/noizeus/airport/0dB/sp01_airport_sn0.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'noizeus', 'dataset_name': 'noizeus', 'lang': 'en', 'subset': 'airport_0dB'}}
3
+ 2026-01-04 13:01:58 | INFO | waiting for other ranks to finish, time elapsed: 10s
4
+ 2026-01-04 13:01:58 | INFO | model Step-Audio-2-mini-all-lora4, data noizeus, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus.jsonl.
5
+ 2026-01-04 13:01:58 | INFO | skip eval for noizeus
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank1.log ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ 2026-01-04 13:00:54 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: noizeus
2
+ 2026-01-04 13:00:54 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets/noizeus/noizeus/airport/0dB/sp02_airport_sn0.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'noizeus', 'dataset_name': 'noizeus', 'lang': 'en', 'subset': 'airport_0dB'}}