Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitignore +184 -0
- .gitmodules +9 -0
- Dockerfile +29 -0
- LEADERBOARD.md +403 -0
- README.md +169 -0
- README_zh.md +168 -0
- config.yaml +7 -0
- constraints.txt +2 -0
- draw_wer.py +69 -0
- ignore-words.txt +2 -0
- log_kimilibri.txt +0 -0
- log_mini_-5_to_10_all_new_noise.txt +0 -0
- log_mini_ea_-5_to_10_all_new_noise.txt +0 -0
- log_mini_ea_-5_to_10_new_noise.txt +0 -0
- log_mini_origin1.txt +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_default_performance.json +25 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_wer_details.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank0.log +8 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank1.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank2.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank3.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank4.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank5.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank6.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank7.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/Step-Audio-2-mini-all-lora3_voices_dev_test_wer_details.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank0.log +15 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank1.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank2.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank3.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank4.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank5.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank6.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_default_performance.json +25 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_wer_details.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank0.log +6 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank1.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank2.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank3.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank4.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank5.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank6.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank7.log +2 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_default_performance.json +121 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_wer_details.jsonl +0 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank0.log +5 -0
- mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank1.log +2 -0
.gitignore
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Byte-compiled / optimized / DLL files
|
| 2 |
+
__pycache__/
|
| 3 |
+
*.py[cod]
|
| 4 |
+
*$py.class
|
| 5 |
+
data/test.sh
|
| 6 |
+
# C extensions
|
| 7 |
+
*.so
|
| 8 |
+
|
| 9 |
+
# Distribution / packaging
|
| 10 |
+
.Python
|
| 11 |
+
build/
|
| 12 |
+
develop-eggs/
|
| 13 |
+
dist/
|
| 14 |
+
downloads/
|
| 15 |
+
eggs/
|
| 16 |
+
.eggs/
|
| 17 |
+
lib/
|
| 18 |
+
lib64/
|
| 19 |
+
parts/
|
| 20 |
+
sdist/
|
| 21 |
+
var/
|
| 22 |
+
wheels/
|
| 23 |
+
share/python-wheels/
|
| 24 |
+
*.egg-info/
|
| 25 |
+
.installed.cfg
|
| 26 |
+
*.egg
|
| 27 |
+
MANIFEST
|
| 28 |
+
|
| 29 |
+
# PyInstaller
|
| 30 |
+
# Usually these files are written by a python_goose script from a template
|
| 31 |
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
| 32 |
+
*.manifest
|
| 33 |
+
*.spec
|
| 34 |
+
|
| 35 |
+
# Installer logs
|
| 36 |
+
pip-log.txt
|
| 37 |
+
pip-delete-this-directory.txt
|
| 38 |
+
|
| 39 |
+
# Unit test / coverage reports
|
| 40 |
+
htmlcov/
|
| 41 |
+
.tox/
|
| 42 |
+
.nox/
|
| 43 |
+
.coverage
|
| 44 |
+
.coverage.*
|
| 45 |
+
.cache
|
| 46 |
+
nosetests.xml
|
| 47 |
+
coverage.xml
|
| 48 |
+
*.cover
|
| 49 |
+
*.py,cover
|
| 50 |
+
.hypothesis/
|
| 51 |
+
.pytest_cache/
|
| 52 |
+
cover/
|
| 53 |
+
|
| 54 |
+
# Translations
|
| 55 |
+
*.mo
|
| 56 |
+
*.pot
|
| 57 |
+
|
| 58 |
+
# Django stuff:
|
| 59 |
+
*.log
|
| 60 |
+
local_settings.py
|
| 61 |
+
db.sqlite3
|
| 62 |
+
db.sqlite3-journal
|
| 63 |
+
|
| 64 |
+
# Flask stuff:
|
| 65 |
+
instance/
|
| 66 |
+
.webassets-cache
|
| 67 |
+
|
| 68 |
+
# Scrapy stuff:
|
| 69 |
+
.scrapy
|
| 70 |
+
|
| 71 |
+
# Sphinx documentation
|
| 72 |
+
docs/_build/
|
| 73 |
+
|
| 74 |
+
# PyBuilder
|
| 75 |
+
.pybuilder/
|
| 76 |
+
target/
|
| 77 |
+
|
| 78 |
+
# Jupyter Notebook
|
| 79 |
+
.ipynb_checkpoints
|
| 80 |
+
|
| 81 |
+
# IPython
|
| 82 |
+
profile_default/
|
| 83 |
+
ipython_config.py
|
| 84 |
+
|
| 85 |
+
# pyenv
|
| 86 |
+
# For a library or package, you might want to ignore these files since the code is
|
| 87 |
+
# intended to run in multiple environments; otherwise, check them in:
|
| 88 |
+
# .python_goose-version
|
| 89 |
+
|
| 90 |
+
# pipenv
|
| 91 |
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
| 92 |
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
| 93 |
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
| 94 |
+
# install all needed dependencies.
|
| 95 |
+
#Pipfile.lock
|
| 96 |
+
|
| 97 |
+
# poetry
|
| 98 |
+
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
| 99 |
+
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
| 100 |
+
# commonly ignored for libraries.
|
| 101 |
+
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
| 102 |
+
#poetry.lock
|
| 103 |
+
|
| 104 |
+
# pdm
|
| 105 |
+
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
| 106 |
+
#pdm.lock
|
| 107 |
+
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
| 108 |
+
# in version control.
|
| 109 |
+
# https://pdm.fming.dev/#use-with-ide
|
| 110 |
+
.pdm.toml
|
| 111 |
+
|
| 112 |
+
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
| 113 |
+
__pypackages__/
|
| 114 |
+
|
| 115 |
+
# Celery stuff
|
| 116 |
+
celerybeat-schedule
|
| 117 |
+
celerybeat.pid
|
| 118 |
+
|
| 119 |
+
# SageMath parsed files
|
| 120 |
+
*.sage.py
|
| 121 |
+
|
| 122 |
+
# Environments
|
| 123 |
+
.env
|
| 124 |
+
.env_*
|
| 125 |
+
.venv
|
| 126 |
+
env/
|
| 127 |
+
venv/
|
| 128 |
+
ENV/
|
| 129 |
+
env.bak/
|
| 130 |
+
venv.bak/
|
| 131 |
+
|
| 132 |
+
# Spyder project settings
|
| 133 |
+
.spyderproject
|
| 134 |
+
.spyproject
|
| 135 |
+
|
| 136 |
+
# Rope project settings
|
| 137 |
+
.ropeproject
|
| 138 |
+
|
| 139 |
+
# mkdocs documentation
|
| 140 |
+
/site
|
| 141 |
+
|
| 142 |
+
# mypy
|
| 143 |
+
.mypy_cache/
|
| 144 |
+
.dmypy.json
|
| 145 |
+
dmypy.json
|
| 146 |
+
|
| 147 |
+
# Pyre type checker
|
| 148 |
+
.pyre/
|
| 149 |
+
|
| 150 |
+
# pytype static type analyzer
|
| 151 |
+
.pytype/
|
| 152 |
+
|
| 153 |
+
# Cython debug symbols
|
| 154 |
+
cython_debug/
|
| 155 |
+
|
| 156 |
+
# PyCharm
|
| 157 |
+
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
| 158 |
+
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
| 159 |
+
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
| 160 |
+
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
| 161 |
+
.idea/
|
| 162 |
+
data/downloaded_datasets/*/*/*.wav
|
| 163 |
+
data/downloaded_datasets/*/*.wav
|
| 164 |
+
data/downloaded_datasets/*.wav
|
| 165 |
+
data/downloaded_datasets
|
| 166 |
+
logs
|
| 167 |
+
*.zip
|
| 168 |
+
conf
|
| 169 |
+
.DS_Store
|
| 170 |
+
.ruff_cache
|
| 171 |
+
.log
|
| 172 |
+
*.jsonl
|
| 173 |
+
*.mp3
|
| 174 |
+
*.json
|
| 175 |
+
*.parquet
|
| 176 |
+
*.progress
|
| 177 |
+
# Vscode
|
| 178 |
+
.vscode
|
| 179 |
+
eval_result/
|
| 180 |
+
debug/
|
| 181 |
+
almeval/models/kimia_hf
|
| 182 |
+
|
| 183 |
+
# .pyc
|
| 184 |
+
*.pyc
|
.gitmodules
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[submodule "almeval/models/glm4voice"]
|
| 2 |
+
path = almeval/models/glm4voice
|
| 3 |
+
url = https://github.com/THUDM/GLM-4-Voice.git
|
| 4 |
+
[submodule "almeval/models/stepaudio"]
|
| 5 |
+
path = almeval/models/stepaudio
|
| 6 |
+
url = https://github.com/stepfun-ai/Step-Audio.git
|
| 7 |
+
[submodule "almeval/models/kimi_audio"]
|
| 8 |
+
path = almeval/models/kimi_audio
|
| 9 |
+
url = https://github.com/MoonshotAI/Kimi-Audio.git
|
Dockerfile
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04
|
| 2 |
+
|
| 3 |
+
WORKDIR /app
|
| 4 |
+
|
| 5 |
+
COPY ./requirements.txt /app/
|
| 6 |
+
RUN apt-get update && apt-get install -y \
|
| 7 |
+
python3.10 \
|
| 8 |
+
python3.10-dev \
|
| 9 |
+
curl \
|
| 10 |
+
sox \
|
| 11 |
+
openssh-server \
|
| 12 |
+
ffmpeg \
|
| 13 |
+
libgl1-mesa-glx \
|
| 14 |
+
git \
|
| 15 |
+
ninja-build \
|
| 16 |
+
&& rm -rf /var/lib/apt/lists/*
|
| 17 |
+
|
| 18 |
+
# 安装 pip
|
| 19 |
+
RUN curl https://bootstrap.pypa.io/get-pip.py -o get-pip.py \
|
| 20 |
+
&& python3.10 get-pip.py \
|
| 21 |
+
&& rm get-pip.py
|
| 22 |
+
RUN pip install -r requirements.txt
|
| 23 |
+
RUN pip install git+https://github.com/huggingface/transformers@v4.51.3-Qwen2.5-Omni-preview
|
| 24 |
+
RUN pip install flash-attn --no-build-isolation
|
| 25 |
+
|
| 26 |
+
# alias python3 as python
|
| 27 |
+
RUN ln -s /usr/bin/python3 /usr/bin/python
|
| 28 |
+
|
| 29 |
+
CMD ["/bin/bash"]
|
LEADERBOARD.md
ADDED
|
@@ -0,0 +1,403 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Leaderboard
|
| 2 |
+
|
| 3 |
+
## Automatic Speech Recognition (ASR)
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
<table>
|
| 7 |
+
|
| 8 |
+
<thead>
|
| 9 |
+
<tr>
|
| 10 |
+
<th>Datasets</th>
|
| 11 |
+
<th>Model</th>
|
| 12 |
+
<th>Performance (WER↓)</th>
|
| 13 |
+
</tr>
|
| 14 |
+
</thead>
|
| 15 |
+
<tbody>
|
| 16 |
+
<tr>
|
| 17 |
+
<td rowspan="5"><b>LibriSpeech</b><br>test-clean | test-other</td>
|
| 18 |
+
<td>Qwen2-Audio-base</td>
|
| 19 |
+
<td>1.74 | 4.04</td>
|
| 20 |
+
</tr>
|
| 21 |
+
<tr>
|
| 22 |
+
<td>Baichuan-base</td>
|
| 23 |
+
<td>3.02 | 6.04</td>
|
| 24 |
+
</tr>
|
| 25 |
+
<tr>
|
| 26 |
+
<td>StepAudio-chat</td>
|
| 27 |
+
<td>3.19 | 10.67</td>
|
| 28 |
+
</tr>
|
| 29 |
+
<tr>
|
| 30 |
+
<td>Qwen2.5-Omni</td>
|
| 31 |
+
<td>2.37 | 4.21</td>
|
| 32 |
+
</tr>
|
| 33 |
+
<tr>
|
| 34 |
+
<td>Kimi-Audio</td>
|
| 35 |
+
<td><b>1.28 | 2.42</b></td>
|
| 36 |
+
</tr>
|
| 37 |
+
<tr>
|
| 38 |
+
<td rowspan="5"><b>Fleurs</b><br>zh | en</td>
|
| 39 |
+
<td>Qwen2-Audio-base</td>
|
| 40 |
+
<td>3.63 | 5.20</td>
|
| 41 |
+
</tr>
|
| 42 |
+
<tr>
|
| 43 |
+
<td>Baichuan-base</td>
|
| 44 |
+
<td>4.15 | 8.07</td>
|
| 45 |
+
</tr>
|
| 46 |
+
<tr>
|
| 47 |
+
<td>StepAudio-chat</td>
|
| 48 |
+
<td>4.26 | 8.56</td>
|
| 49 |
+
</tr>
|
| 50 |
+
<tr>
|
| 51 |
+
<td>Qwen2.5-Omni</td>
|
| 52 |
+
<td>2.92 | <b>4.17</b></td>
|
| 53 |
+
</tr>
|
| 54 |
+
<tr>
|
| 55 |
+
<td>Kimi-Audio</td>
|
| 56 |
+
<td><b>2.69</b> | 4.44</td>
|
| 57 |
+
</tr>
|
| 58 |
+
<tr>
|
| 59 |
+
<td rowspan="5"><b>AISHELL-1</b></td>
|
| 60 |
+
<td>Qwen2-Audio-base</td>
|
| 61 |
+
<td>1.52</td>
|
| 62 |
+
</tr>
|
| 63 |
+
<tr>
|
| 64 |
+
<td>Baichuan-base</td>
|
| 65 |
+
<td>1.93</td>
|
| 66 |
+
</tr>
|
| 67 |
+
<tr>
|
| 68 |
+
<td>StepAudio-chat</td>
|
| 69 |
+
<td>2.14</td>
|
| 70 |
+
</tr>
|
| 71 |
+
<tr>
|
| 72 |
+
<td>Qwen2.5-Omni</td>
|
| 73 |
+
<td>1.13</td>
|
| 74 |
+
</tr>
|
| 75 |
+
<tr>
|
| 76 |
+
<td>Kimi-Audio</td>
|
| 77 |
+
<td><b>0.60</b></td>
|
| 78 |
+
</tr>
|
| 79 |
+
<tr>
|
| 80 |
+
<td rowspan="5"><b>AISHELL-2</b> ios</td>
|
| 81 |
+
<td>Qwen2-Audio-base</td>
|
| 82 |
+
<td>3.08</td>
|
| 83 |
+
</tr>
|
| 84 |
+
<tr>
|
| 85 |
+
<td>Baichuan-base</td>
|
| 86 |
+
<td>3.87</td>
|
| 87 |
+
</tr>
|
| 88 |
+
<tr>
|
| 89 |
+
<td>StepAudio-chat</td>
|
| 90 |
+
<td>3.89</td>
|
| 91 |
+
</tr>
|
| 92 |
+
<tr>
|
| 93 |
+
<td>Qwen2.5-Omni</td>
|
| 94 |
+
<td><b>2.56</b></td>
|
| 95 |
+
</tr>
|
| 96 |
+
<tr>
|
| 97 |
+
<td>Kimi-Audio</td>
|
| 98 |
+
<td><b>2.56</b></td>
|
| 99 |
+
</tr>
|
| 100 |
+
<tr>
|
| 101 |
+
<td rowspan="5"><b>WenetSpeech</b><br>test-meeting | test-net</td>
|
| 102 |
+
<td>Qwen2-Audio-base</td>
|
| 103 |
+
<td>8.40 | 7.64</td>
|
| 104 |
+
</tr>
|
| 105 |
+
<tr>
|
| 106 |
+
<td>Baichuan-base</td>
|
| 107 |
+
<td>13.28 | 10.13</td>
|
| 108 |
+
</tr>
|
| 109 |
+
<tr>
|
| 110 |
+
<td>StepAudio-chat</td>
|
| 111 |
+
<td>10.83 | 9.47</td>
|
| 112 |
+
</tr>
|
| 113 |
+
<tr>
|
| 114 |
+
<td>Qwen2.5-Omni</td>
|
| 115 |
+
<td>7.71 | 6.04</td>
|
| 116 |
+
</tr>
|
| 117 |
+
<tr>
|
| 118 |
+
<td>Kimi-Audio</td>
|
| 119 |
+
<td><b>6.28 | 5.37</b></td>
|
| 120 |
+
</tr>
|
| 121 |
+
<tr>
|
| 122 |
+
<td rowspan="5"><b>Kimi-ASR Internal Testset</b><br>subset1 | subset2</td>
|
| 123 |
+
<td>Qwen2-Audio-base</td>
|
| 124 |
+
<td>2.31 | 3.24</td>
|
| 125 |
+
</tr>
|
| 126 |
+
<tr>
|
| 127 |
+
<td>Baichuan-base</td>
|
| 128 |
+
<td>3.41 | 5.60</td>
|
| 129 |
+
</tr>
|
| 130 |
+
<tr>
|
| 131 |
+
<td>StepAudio-chat</td>
|
| 132 |
+
<td>2.82 | 4.74</td>
|
| 133 |
+
</tr>
|
| 134 |
+
<tr>
|
| 135 |
+
<td>Qwen2.5-Omni</td>
|
| 136 |
+
<td>1.53 | 2.68</td>
|
| 137 |
+
</tr>
|
| 138 |
+
<tr>
|
| 139 |
+
<td>Kimi-Audio</td>
|
| 140 |
+
<td><b>1.42 | 2.44</b></td>
|
| 141 |
+
</tr>
|
| 142 |
+
</tbody>
|
| 143 |
+
</table>
|
| 144 |
+
|
| 145 |
+
## Audio Understanding
|
| 146 |
+
|
| 147 |
+
<table>
|
| 148 |
+
<thead>
|
| 149 |
+
<tr>
|
| 150 |
+
<th>Datasets</th>
|
| 151 |
+
<th>Model</th>
|
| 152 |
+
<th>Performance↑</th>
|
| 153 |
+
</tr>
|
| 154 |
+
</thead>
|
| 155 |
+
<tbody>
|
| 156 |
+
<tr>
|
| 157 |
+
<td rowspan="6"><b>MMAU</b><br>music | sound | speech</td>
|
| 158 |
+
<td>Qwen2-Audio-base</td>
|
| 159 |
+
<td>58.98 | 69.07 | 52.55</td>
|
| 160 |
+
</tr>
|
| 161 |
+
<tr>
|
| 162 |
+
<td>Baichuan-chat</td>
|
| 163 |
+
<td>49.10 | 59.46 | 42.47</td>
|
| 164 |
+
</tr>
|
| 165 |
+
<tr>
|
| 166 |
+
<td>GLM-4-Voice</td>
|
| 167 |
+
<td>38.92 | 43.54 | 32.43</td>
|
| 168 |
+
</tr>
|
| 169 |
+
<tr>
|
| 170 |
+
<td>StepAudio-chat</td>
|
| 171 |
+
<td>49.40 | 53.75 | 47.75</td>
|
| 172 |
+
</tr>
|
| 173 |
+
<tr>
|
| 174 |
+
<td>Qwen2.5-Omni</td>
|
| 175 |
+
<td><b>62.16</b> | 67.57 | 53.92</td>
|
| 176 |
+
</tr>
|
| 177 |
+
<tr>
|
| 178 |
+
<td>Kimi-Audio</td>
|
| 179 |
+
<td>61.68 | <b>73.27</b> | <b>60.66</b></td>
|
| 180 |
+
</tr>
|
| 181 |
+
<tr>
|
| 182 |
+
<td rowspan="5"><b>ClothoAQA</b><br>test | dev</td>
|
| 183 |
+
<td>Qwen2-Audio-base</td>
|
| 184 |
+
<td>71.73 | 72.63</td>
|
| 185 |
+
</tr>
|
| 186 |
+
<tr>
|
| 187 |
+
<td>Baichuan-chat</td>
|
| 188 |
+
<td>48.02 | 48.16</td>
|
| 189 |
+
</tr>
|
| 190 |
+
<tr>
|
| 191 |
+
<td>StepAudio-chat</td>
|
| 192 |
+
<td>45.84 | 44.98</td>
|
| 193 |
+
</tr>
|
| 194 |
+
<tr>
|
| 195 |
+
<td>Qwen2.5-Omni</td>
|
| 196 |
+
<td><b>72.86</b> | 73.12</td>
|
| 197 |
+
</tr>
|
| 198 |
+
<tr>
|
| 199 |
+
<td>Kimi-Audio</td>
|
| 200 |
+
<td>71.24 | <b>73.18</b></td>
|
| 201 |
+
</tr>
|
| 202 |
+
<tr>
|
| 203 |
+
<td rowspan="5"><b>VocalSound</b></td>
|
| 204 |
+
<td>Qwen2-Audio-base</td>
|
| 205 |
+
<td>93.82</td>
|
| 206 |
+
</tr>
|
| 207 |
+
<tr>
|
| 208 |
+
<td>Baichuan-base</td>
|
| 209 |
+
<td>58.17</td>
|
| 210 |
+
</tr>
|
| 211 |
+
<tr>
|
| 212 |
+
<td>StepAudio-chat</td>
|
| 213 |
+
<td>28.58</td>
|
| 214 |
+
</tr>
|
| 215 |
+
<tr>
|
| 216 |
+
<td>Qwen2.5-Omni</td>
|
| 217 |
+
<td>93.73</td>
|
| 218 |
+
</tr>
|
| 219 |
+
<tr>
|
| 220 |
+
<td>Kimi-Audio</td>
|
| 221 |
+
<td><b>94.85</b></td>
|
| 222 |
+
</tr>
|
| 223 |
+
<tr>
|
| 224 |
+
<td rowspan="5"><b>Nonspeech7k</b></td>
|
| 225 |
+
<td>Qwen2-Audio-base</td>
|
| 226 |
+
<td>87.17</td>
|
| 227 |
+
</tr>
|
| 228 |
+
<tr>
|
| 229 |
+
<td>Baichuan-chat</td>
|
| 230 |
+
<td>59.03</td>
|
| 231 |
+
</tr>
|
| 232 |
+
<tr>
|
| 233 |
+
<td>StepAudio-chat</td>
|
| 234 |
+
<td>21.38</td>
|
| 235 |
+
</tr>
|
| 236 |
+
<tr>
|
| 237 |
+
<td>Qwen2.5-Omni</td>
|
| 238 |
+
<td>69.89</td>
|
| 239 |
+
</tr>
|
| 240 |
+
<tr>
|
| 241 |
+
<td>Kimi-Audio</td>
|
| 242 |
+
<td><b>93.93</b></td>
|
| 243 |
+
</tr>
|
| 244 |
+
<tr>
|
| 245 |
+
<td rowspan="5"><b>MELD</b></td>
|
| 246 |
+
<td>Qwen2-Audio-base</td>
|
| 247 |
+
<td>51.23</td>
|
| 248 |
+
</tr>
|
| 249 |
+
<tr>
|
| 250 |
+
<td>Baichuan-chat</td>
|
| 251 |
+
<td>23.59</td>
|
| 252 |
+
</tr>
|
| 253 |
+
<tr>
|
| 254 |
+
<td>StepAudio-chat</td>
|
| 255 |
+
<td>33.54</td>
|
| 256 |
+
</tr>
|
| 257 |
+
<tr>
|
| 258 |
+
<td>Qwen2.5-Omni</td>
|
| 259 |
+
<td>49.83</td>
|
| 260 |
+
</tr>
|
| 261 |
+
<tr>
|
| 262 |
+
<td>Kimi-Audio</td>
|
| 263 |
+
<td><b>59.13</b></td>
|
| 264 |
+
</tr>
|
| 265 |
+
<tr>
|
| 266 |
+
<td rowspan="5"><b>TUT2017</b></td>
|
| 267 |
+
<td>Qwen2-Audio-base</td>
|
| 268 |
+
<td>33.83</td>
|
| 269 |
+
</tr>
|
| 270 |
+
<tr>
|
| 271 |
+
<td>Baichuan-base</td>
|
| 272 |
+
<td>27.9</td>
|
| 273 |
+
</tr>
|
| 274 |
+
<tr>
|
| 275 |
+
<td>StepAudio-chat</td>
|
| 276 |
+
<td>7.41</td>
|
| 277 |
+
</tr>
|
| 278 |
+
<tr>
|
| 279 |
+
<td>Qwen2.5-Omni</td>
|
| 280 |
+
<td>43.27</td>
|
| 281 |
+
</tr>
|
| 282 |
+
<tr>
|
| 283 |
+
<td>Kimi-Audio</td>
|
| 284 |
+
<td><b>65.25</b></td>
|
| 285 |
+
</tr>
|
| 286 |
+
<tr>
|
| 287 |
+
<td rowspan="5"><b>CochlScene</b><br>test | dev</td>
|
| 288 |
+
<td>Qwen2-Audio-base</td>
|
| 289 |
+
<td>52.69 | 50.96</td>
|
| 290 |
+
</tr>
|
| 291 |
+
<tr>
|
| 292 |
+
<td>Baichuan-base</td>
|
| 293 |
+
<td>34.93 | 34.56</td>
|
| 294 |
+
</tr>
|
| 295 |
+
<tr>
|
| 296 |
+
<td>StepAudio-chat</td>
|
| 297 |
+
<td>10.06 | 10.42</td>
|
| 298 |
+
</tr>
|
| 299 |
+
<tr>
|
| 300 |
+
<td>Qwen2.5-Omni</td>
|
| 301 |
+
<td>63.82 | 63.82</td>
|
| 302 |
+
</tr>
|
| 303 |
+
<tr>
|
| 304 |
+
<td>Kimi-Audio</td>
|
| 305 |
+
<td><b>79.84 | 80.99</b></td>
|
| 306 |
+
</tr>
|
| 307 |
+
</tbody>
|
| 308 |
+
</table>
|
| 309 |
+
|
| 310 |
+
## Audio-to-Text Chat
|
| 311 |
+
|
| 312 |
+
<table>
|
| 313 |
+
<thead>
|
| 314 |
+
<tr>
|
| 315 |
+
<th>Datasets</th>
|
| 316 |
+
<th>Model</th>
|
| 317 |
+
<th>Performance↑</th>
|
| 318 |
+
</tr>
|
| 319 |
+
</thead>
|
| 320 |
+
<tbody>
|
| 321 |
+
<tr>
|
| 322 |
+
<td rowspan="6"><b>OpenAudioBench</b><br>AlpacaEval | Llama Questions |<br>Reasoning QA | TriviaQA | Web Questions</td>
|
| 323 |
+
<td>Qwen2-Audio-chat</td>
|
| 324 |
+
<td>57.19 | 69.67 | 42.77 | 40.30 | 45.20</td>
|
| 325 |
+
</tr>
|
| 326 |
+
<tr>
|
| 327 |
+
<td>Baichuan-chat</td>
|
| 328 |
+
<td>59.65 | 74.33 | 46.73 | 55.40 | 58.70</td>
|
| 329 |
+
</tr>
|
| 330 |
+
<tr>
|
| 331 |
+
<td>GLM-4-Voice</td>
|
| 332 |
+
<td>57.89 | 76.00 | 47.43 | 51.80 | 55.40</td>
|
| 333 |
+
</tr>
|
| 334 |
+
<tr>
|
| 335 |
+
<td>StepAudio-chat</td>
|
| 336 |
+
<td>56.53 | 72.33 | 60.00 | 56.80 | <b>73.00</b></td>
|
| 337 |
+
</tr>
|
| 338 |
+
<tr>
|
| 339 |
+
<td>Qwen2.5-Omni</td>
|
| 340 |
+
<td>72.76 | 75.33 | <b>63.76</b> | 57.06 | 62.80</td>
|
| 341 |
+
</tr>
|
| 342 |
+
<tr>
|
| 343 |
+
<td>Kimi-Audio</td>
|
| 344 |
+
<td><b>75.73</b> | <b>79.33</b> | 58.02 | <b>62.10 </b> | 70.20</td>
|
| 345 |
+
</tr>
|
| 346 |
+
<tr>
|
| 347 |
+
<td rowspan="6"><b>VoiceBench</b><br>AlpacaEval | CommonEval |<br>SD-QA | MMSU</td>
|
| 348 |
+
<td>Qwen2-Audio-chat</td>
|
| 349 |
+
<td>3.69 | 3.40 | 35.35 | 35.43</td>
|
| 350 |
+
</tr>
|
| 351 |
+
<tr>
|
| 352 |
+
<td>Baichuan-chat</td>
|
| 353 |
+
<td>4.00 | 3.39 | 49.64 | 48.80</td>
|
| 354 |
+
</tr>
|
| 355 |
+
<tr>
|
| 356 |
+
<td>GLM-4-Voice</td>
|
| 357 |
+
<td>4.06 | 3.48 | 43.31 | 40.11</td>
|
| 358 |
+
</tr>
|
| 359 |
+
<tr>
|
| 360 |
+
<td>StepAudio-chat</td>
|
| 361 |
+
<td>3.99 | 2.99 | 46.84 | 28.72</td>
|
| 362 |
+
</tr>
|
| 363 |
+
<tr>
|
| 364 |
+
<td>Qwen2.5-Omni</td>
|
| 365 |
+
<td>4.33 | 3.84 | 57.41 | 56.38</td>
|
| 366 |
+
</tr>
|
| 367 |
+
<tr>
|
| 368 |
+
<td>Kimi-Audio</td>
|
| 369 |
+
<td><b>4.46</b> | <b>3.97</b> | <b>63.12</b> | <b>62.17</b></td>
|
| 370 |
+
</tr>
|
| 371 |
+
<tr>
|
| 372 |
+
<td rowspan="6"><b>VoiceBench</b><br>OpenBookQA | IFEval |<br>AdvBench | Avg</td>
|
| 373 |
+
<td>Qwen2-Audio-chat</td>
|
| 374 |
+
<td>49.01 | 22.57 | 98.85 | 54.72</td>
|
| 375 |
+
</tr>
|
| 376 |
+
<tr>
|
| 377 |
+
<td>Baichuan-chat</td>
|
| 378 |
+
<td>63.30 | 41.32 | 86.73 | 62.51</td>
|
| 379 |
+
</tr>
|
| 380 |
+
<tr>
|
| 381 |
+
<td>GLM-4-Voice</td>
|
| 382 |
+
<td>52.97 | 24.91 | 88.08 | 57.17</td>
|
| 383 |
+
</tr>
|
| 384 |
+
<tr>
|
| 385 |
+
<td>StepAudio-chat</td>
|
| 386 |
+
<td>31.87 | 29.19 | 65.77 | 48.86</td>
|
| 387 |
+
</tr>
|
| 388 |
+
<tr>
|
| 389 |
+
<td>Qwen2.5-Omni</td>
|
| 390 |
+
<td>79.12 | 53.88 | 99.62 | 72.83</td>
|
| 391 |
+
</tr>
|
| 392 |
+
<tr>
|
| 393 |
+
<td>Kimi-Audio</td>
|
| 394 |
+
<td><b>83.52</b> | <b>61.10</b> | <b>100.00</b> | <b>76.93</b></td>
|
| 395 |
+
</tr>
|
| 396 |
+
</tbody>
|
| 397 |
+
</table>
|
| 398 |
+
|
| 399 |
+
|
| 400 |
+
|
| 401 |
+
## Updates
|
| 402 |
+
- 2025-04-25: Initial leaderboard created
|
| 403 |
+
|
README.md
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Kimi-Audio-Evalkit
|
| 2 |
+
|
| 3 |
+
[中文版本](README_zh.md)
|
| 4 |
+
|
| 5 |
+
## Introduction
|
| 6 |
+
|
| 7 |
+
Kimi-Audio-Evalkit is an evaluation framework designed for audio large language models. Based on Kimi-Audio-Evalkit, you can quickly implement your own models or datasets and conduct fair comparisons with other open-source models.
|
| 8 |
+
|
| 9 |
+
Our work [Kimi-Audio](https://github.com/MoonshotAI/Kimi-Audio-Evalkit) is evaluated using this framework.
|
| 10 |
+
|
| 11 |
+
See [Leaderboard](./LEADERBOARD.md) for current results.
|
| 12 |
+
|
| 13 |
+
## Getting Started
|
| 14 |
+
|
| 15 |
+
### Step1: Get the Code
|
| 16 |
+
|
| 17 |
+
```bash
|
| 18 |
+
git clone https://github.com/MoonshotAI/Kimi-Audio-Evalkit.git
|
| 19 |
+
cd Kimi-Audio-Evalkit
|
| 20 |
+
git submodule update --init --recursive
|
| 21 |
+
```
|
| 22 |
+
|
| 23 |
+
### Step2: Prepare Environment
|
| 24 |
+
|
| 25 |
+
You can directly use our pre-built Docker image. If you need to update the environment, you can modify the Dockerfile and rebuild it.
|
| 26 |
+
```bash
|
| 27 |
+
docker pull moonshotai/almevalkit:v0.4
|
| 28 |
+
```
|
| 29 |
+
Typically, you need to mount a local directory as the workspace to ensure evaluation results persist after container exit:
|
| 30 |
+
```bash
|
| 31 |
+
docker run -it -v $(pwd):/app moonshotai/almevalkit:v0.4 bash
|
| 32 |
+
```
|
| 33 |
+
|
| 34 |
+
### Step3: Get Datasets
|
| 35 |
+
|
| 36 |
+
Most datasets used by ALMEvalKit can be downloaded using our included tools. Some datasets cannot be fully automated. Please refer to [Download Datasets](./data/README.md) for details.
|
| 37 |
+
For datasets on Hugging Face, we will soon provide a more direct usage method. Please stay tuned for updates.
|
| 38 |
+
|
| 39 |
+
### Step4: Configure config.yaml
|
| 40 |
+
You may need to fill in several fields in the config.yaml in the root directory to help us locate your data source. By default, datasets will be downloaded to the data/ directory under the current directory. If you downloaded them elsewhere, please enter the root directory in the dataset_root field.
|
| 41 |
+
```yaml
|
| 42 |
+
DATASETS:
|
| 43 |
+
dataset_root: "/path/to/your/dataset/root"
|
| 44 |
+
```
|
| 45 |
+
### Step5: Evaluation
|
| 46 |
+
|
| 47 |
+
run_audio.sh is the entry point for evaluation. You can get help using `--help`
|
| 48 |
+
|
| 49 |
+
For example, to evaluate Kimi-Audio on all datasets:
|
| 50 |
+
```
|
| 51 |
+
bash run_audio.sh --model Kimi-Audio --data all --skip-eval
|
| 52 |
+
```
|
| 53 |
+
By default, inference results, evaluation results, and metric reports will be generated in the eval_results directory under the current directory. You can change this behavior by passing --work-dir.
|
| 54 |
+
|
| 55 |
+
Using --skip-eval allows the model to only perform inference without evaluation, which helps keep your GPU running efficiently.
|
| 56 |
+
After inference is complete, you can run the command again to start evaluation. You can add the --reeval parameter to force re-evaluation of the dataset, which won't trigger re-inference but will regenerate the metric report.
|
| 57 |
+
|
| 58 |
+
Note: Our default LLM method is gpt-4o-mini. You need to set your own API KEY to enable it. We will support more evaluation models in the future.
|
| 59 |
+
```
|
| 60 |
+
export OPENAI_API_KEY=your_api_key
|
| 61 |
+
bash run_audio.sh --model Kimi-Audio --data all --reeval
|
| 62 |
+
```
|
| 63 |
+
|
| 64 |
+
Currently supported models, datasets, and evaluation models are listed below:
|
| 65 |
+
|
| 66 |
+
**Models**
|
| 67 |
+
|
| 68 |
+
- **Baichuan Series**: Baichuan-Audio-Base, Baichuan-Audio-Instruct
|
| 69 |
+
- **Qwen Series**: Qwen2-Audio-7B, Qwen2-Audio-7B-Instruct, Qwen2.5-Omni-7B
|
| 70 |
+
- **GLM Series**: GLM4-Voice
|
| 71 |
+
- **Others**: StepAudio, Kimi-Audio
|
| 72 |
+
|
| 73 |
+
**Datasets**
|
| 74 |
+
|
| 75 |
+
| Dataset Category | Datasets |
|
| 76 |
+
|-----------------|----------|
|
| 77 |
+
| ASR | LibriSpeech, Fleurs-zh, Fleurs-en, AISHELL-1, AISHELL-2, WenetSpeech |
|
| 78 |
+
| MQA | mmau-test-mini, openbookqa, mmsu, MELD, Nonspeech7k, TUT2017, VocalSound, CochlScene |
|
| 79 |
+
| OpenQA | alpacaeval_full, commoneval, advbench, ifeval |
|
| 80 |
+
| RefQA | ClothoAQA, sd-qa, OpenAudioBench |
|
| 81 |
+
|
| 82 |
+
- For more information about dataset types, ownership, etc., please check the implementation of the relevant datasets.
|
| 83 |
+
|
| 84 |
+
## Adding Datasets
|
| 85 |
+
|
| 86 |
+
We believe the greatest value of ALMEvalKit is not in reproducing existing results, but in providing a simple mechanism to help users add their own datasets and models, and conduct fair comparisons with other model results.
|
| 87 |
+
|
| 88 |
+
We strongly recommend first reading [Dataset Definition](./almeval/datasets/base.py) to understand how we classify datasets. This will help you correctly set the meta information for new datasets, ensuring they are used appropriately.
|
| 89 |
+
|
| 90 |
+
To add a dataset, you need to write a few lines of code to prepare a jsonl file named dataset_name.jsonl for ALMEvalKit. Each line of the jsonl is a json record, and we require each line to have the following fields:
|
| 91 |
+
```
|
| 92 |
+
{
|
| 93 |
+
"index": int, # unique identifier for a piece of data
|
| 94 |
+
"audio_path": str | list[str], # audio location
|
| 95 |
+
"question": str, # question or instruction for the audio, e.g., "Please transcribe the audio content into text". Set to empty if not needed
|
| 96 |
+
"answer": str, # ground truth answer. Set to empty if not needed (e.g., for Open-QA)
|
| 97 |
+
"subset": str, # subset name. Sometimes a dataset can be split into several subsets, which will be evaluated separately and reported independently. If you don't have subsets, use the dataset name
|
| 98 |
+
}
|
| 99 |
+
|
| 100 |
+
For Audio-QA type datasets, we require an additional "audio_content" field to provide the content in text form for LLM evaluation models to assess answer correctness:
|
| 101 |
+
{
|
| 102 |
+
"index": int, # unique identifier for a piece of data
|
| 103 |
+
"audio_path": str | list[str], # audio location
|
| 104 |
+
"question": str, # question or instruction for the audio
|
| 105 |
+
"audio_content": str, # text form of the audio
|
| 106 |
+
"answer": str, # ground truth answer
|
| 107 |
+
"subset": str, # subset name
|
| 108 |
+
}
|
| 109 |
+
```
|
| 110 |
+
[Download Dataset](./data/download_benchmark.py) shows how we download & process data, which you can refer to.
|
| 111 |
+
|
| 112 |
+
After completing this file, you can add your dataset to the appropriate category. Generally, by inheriting the parent class of that category and filling in some fields, your dataset will be ready to use. For example:
|
| 113 |
+
```
|
| 114 |
+
class Vocalsound(AudioMQADataset):
|
| 115 |
+
DATASET_NAME = 'VocalSound'
|
| 116 |
+
DATASET_SERIES = 'VocalSound'
|
| 117 |
+
AUDIO_TYPE = 'AudioEvent'
|
| 118 |
+
```
|
| 119 |
+
This indicates that the vocalsound dataset is an MQA dataset (multiple-choice questions), it belongs to the vocalsound dataset series, its AUDIO_TYPE is marked as "AudioEvent", indicating that this dataset is related to sound events (non-speech), which will affect some model evaluation behaviors during evaluation.
|
| 120 |
+
|
| 121 |
+
If you download it to the dataset cache directory, you can now evaluate this dataset on any model:
|
| 122 |
+
```bash
|
| 123 |
+
bash run_audio.sh --model Kimi-Audio --data vocalsound
|
| 124 |
+
```
|
| 125 |
+
If you saved this file elsewhere, please tell us in config.yaml:
|
| 126 |
+
```yaml
|
| 127 |
+
DATASETS:
|
| 128 |
+
dataset_root: "/path/to/your/dataset/root"
|
| 129 |
+
datasets:
|
| 130 |
+
#example:
|
| 131 |
+
example: "/path/to/your/dataset/example.jsonl"
|
| 132 |
+
Vocalsound: "/path/to/your/dataset/VocalSound.jsonl"
|
| 133 |
+
```
|
| 134 |
+
|
| 135 |
+
## Adding Models
|
| 136 |
+
|
| 137 |
+
Evaluating your model in ALMEvalKit is also very easy. You only need to implement the generate_inner method, whose signature is:
|
| 138 |
+
|
| 139 |
+
```
|
| 140 |
+
def generate_inner(self, msg:dict) -> (str, str)
|
| 141 |
+
```
|
| 142 |
+
**msg** is a piece of data from the dataset, with the following format:
|
| 143 |
+
```python
|
| 144 |
+
{
|
| 145 |
+
"index": int, # the index of a piece of data in the dataset, as above
|
| 146 |
+
"audio": list[str] # in most cases, the length is 1, and audio[0] can be used to get the audio for this piece of data
|
| 147 |
+
"text": str # the "question" field of the data, may be empty
|
| 148 |
+
"meta": dict # dataset meta information, such as audio_type, name, task, etc. If a piece of data has a meta field, it will also be included here
|
| 149 |
+
}
|
| 150 |
+
```
|
| 151 |
+
|
| 152 |
+
This function returns prompt:str, result:str, where prompt is the actual text sent to the model for inference, and result is the model's inference result.
|
| 153 |
+
|
| 154 |
+
**Note** "The actual text sent to the model for inference" is not necessarily equal to `msg['text']`, as we can set rules at runtime to modify it. Usually, we implement a `get_prompt(msg) -> text` to handle this.
|
| 155 |
+
|
| 156 |
+
**The best way to add a model is to copy an already implemented model and follow its pattern**
|
| 157 |
+
|
| 158 |
+
## Call for Contribution
|
| 159 |
+
|
| 160 |
+
We hope the community can work together to build a fair, efficient, and unified audio large language model evaluation framework in the following aspects:
|
| 161 |
+
|
| 162 |
+
- Add features, fix bugs, improve code quality and usability
|
| 163 |
+
- Support more models and datasets
|
| 164 |
+
- Improve readability, contribute examples and docs
|
| 165 |
+
|
| 166 |
+
Due to limited information, we cannot find the best prompts for each model across different tasks/datasets. We also welcome the community to provide best practices, making our leaderboard better reflect the model's true capabilities.
|
| 167 |
+
|
| 168 |
+
We also recommend that you use pre-commit hooks to auto-format your code. see [pre-commit](https://pre-commit.com/)
|
| 169 |
+
|
README_zh.md
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Kimi-Audio-Evalkit
|
| 2 |
+
|
| 3 |
+
[English Version](README.md)
|
| 4 |
+
|
| 5 |
+
## 介绍
|
| 6 |
+
|
| 7 |
+
Kimi-Audio-Evalkit是一个为音频大模型评测设计的评测框架,基于Kimi-Audio-Evalkit,你可以快速实现自己的模型或数据集,并公平的与其他开源模型进行比对。
|
| 8 |
+
|
| 9 |
+
我们的工作[Kimi-Audio](https://github.com/MoonshotAI/Kimi-Audio-Evalkit)基于此框架评测。
|
| 10 |
+
|
| 11 |
+
[Leaderboard](./LEADERBOARD.md)是目前的评测结果。
|
| 12 |
+
|
| 13 |
+
## 开始评测
|
| 14 |
+
|
| 15 |
+
### Step1: 获取代码
|
| 16 |
+
|
| 17 |
+
```bash
|
| 18 |
+
git clone https://github.com/MoonshotAI/Kimi-Audio-Evalkit.git
|
| 19 |
+
cd Kimi-Audio-Evalkit
|
| 20 |
+
git submodule update --init --recursive
|
| 21 |
+
```
|
| 22 |
+
|
| 23 |
+
### Step2: 准备环境
|
| 24 |
+
|
| 25 |
+
你可以直接使用我们预先build好的镜像,如果你需要更新镜像环境,可以修改Dockerfile后重新构造
|
| 26 |
+
```bash
|
| 27 |
+
docker pull moonshotai/almevalkit:v0.4
|
| 28 |
+
```
|
| 29 |
+
通常情况你需要mount本地目录,并将其作为工作目录,以便评测结果在容器退出后仍然存在
|
| 30 |
+
```bash
|
| 31 |
+
docker run -it -v $(pwd):/app moonshotai/almevalkit:v0.4 bash
|
| 32 |
+
```
|
| 33 |
+
|
| 34 |
+
### Step3: 获取数据集
|
| 35 |
+
|
| 36 |
+
ALMEvalKit所用的大部分数据集都可以通过我们附带的工具下载,有些数据集不能全自动执行,具体请参阅[下载数据集](./data/README.md)
|
| 37 |
+
对位于huggingface的数据集,我们很快将提供更直接的使用方式,请关注更新。
|
| 38 |
+
|
| 39 |
+
### Step4: 配置config.yaml
|
| 40 |
+
你也许需要填写根目录下的config.yaml中的若干字段,帮助我们找到你的数据源。默认情况下,数据集会被下载到当前目录的data/downloaded_datasets下,如果你下载到了其他地方,请将根目录填入dataset_root字段。
|
| 41 |
+
```yaml
|
| 42 |
+
DATASETS:
|
| 43 |
+
dataset_root: "/path/to/your/dataset/root"
|
| 44 |
+
```
|
| 45 |
+
|
| 46 |
+
### Step5: 评测
|
| 47 |
+
|
| 48 |
+
run_audio.sh为评测入口,你可以通过`--help`取得帮助
|
| 49 |
+
|
| 50 |
+
例如,我们希望跑Kimi-Audio在全部数据集上的结果:
|
| 51 |
+
```
|
| 52 |
+
bash run_audio.sh --model Kimi-Audio --data all --skip-eval
|
| 53 |
+
```
|
| 54 |
+
默认情况下,推理结果文件、评测结果文件、指标报告文件将生成在当前目录的eval_results目录下,你可以通过传递--work-dir改变这一行为。
|
| 55 |
+
|
| 56 |
+
使用--skip-eval可以让模型只推理,不评测,这样有助于保持你的GPU高效运转。
|
| 57 |
+
推理完毕后,你只需要重新运行一次,即可展开评测,你可以通过添加--reeval参数来强制对数据集重新评测,这不会触发重新推理,但会重新生成指标报告。
|
| 58 |
+
|
| 59 |
+
Note: 我们默认的LLM方式是gpt-4o-mini,你需要设定你自己的API KEY来启用。未来我们将支持更多评测模型。
|
| 60 |
+
```
|
| 61 |
+
export OPENAI_API_KEY=your_api_key
|
| 62 |
+
bash run_audio.sh --model Kimi-Audio --data all --reeval
|
| 63 |
+
```
|
| 64 |
+
|
| 65 |
+
目前已经支持的模型、数据集和评测模型列表如下
|
| 66 |
+
|
| 67 |
+
**模型**
|
| 68 |
+
|
| 69 |
+
- **Baichuan Series**: Baichuan-Audio-Base, Baichuan-Audio-Instruct
|
| 70 |
+
- **Qwen Series**: Qwen2-Audio-7B, Qwen2-Audio-7B-Instruct, Qwen2.5-Omni-7B
|
| 71 |
+
- **GLM Series**: GLM4-Voice
|
| 72 |
+
- **Others**: StepAudio, Kimi-Audio
|
| 73 |
+
|
| 74 |
+
**数据集**
|
| 75 |
+
| 数据集类型 | 数据集 |
|
| 76 |
+
|-----------------|----------|
|
| 77 |
+
| ASR | LibriSpeech, Fleurs-zh, Fleurs-en, AISHELL-1, AISHELL-2, WenetSpeech |
|
| 78 |
+
| MQA | mmau-test-mini, openbookqa, mmsu, MELD, Nonspeech7k, TUT2017, VocalSound, CochlScene |
|
| 79 |
+
| OpenQA | alpacaeval_full, commoneval, advbench, ifeval |
|
| 80 |
+
| RefQA | ClothoAQA, sd-qa, OpenAudioBench |
|
| 81 |
+
|
| 82 |
+
- 数据集的类型、归属等更多信息,可以查看相关数据集的实现。
|
| 83 |
+
|
| 84 |
+
## 添加数据集
|
| 85 |
+
|
| 86 |
+
我们相信ALMEvalKit的最大价值不是复现某个已有结果,而是提供一种简单的机制帮助用户添加自己的数据集和模型,并能够与其他模型结果公平比较。
|
| 87 |
+
|
| 88 |
+
我们强烈建议首先阅读[数据集的定义](./almeval/datasets/base.py)了解我们如何对数据集分类,这将帮助你正确的设定新数据集的meta信息,使它们被更正确的使用。
|
| 89 |
+
|
| 90 |
+
要添加数据集,你需要写几行代码,为ALMEvalKit准备一个名为dataset_name.jsonl的jsonl文件。jsonl的每一行是一个json记录,我们要求每一行必须具有的字段是:
|
| 91 |
+
```
|
| 92 |
+
{
|
| 93 |
+
"index": int, # 一条数据的唯一标识
|
| 94 |
+
"audio_path": str | list[str], # 音频位置
|
| 95 |
+
"question": str, # 针对音频的问题或指令,例如"请将音频内容转写为文字",如果你不需要此字段,请设为空
|
| 96 |
+
"answer": str, # ground truth答案,如果你不需要此字段(如Open-QA),请设为空
|
| 97 |
+
"subset": str, # 子数据,有时候一个数据集可以被切分为若干个子集,这些子集将被分别评估,独立汇报结果。如果你没有子数据集,填数据集名字即可
|
| 98 |
+
}
|
| 99 |
+
|
| 100 |
+
对于Audio-QA类的数据集,我们要求额外增加一个"audio_content"字段,以文字形式写出内容,以便交给LLM评测模型评测答案是否正确。
|
| 101 |
+
{
|
| 102 |
+
"index": int, # 一条数据的唯一标识
|
| 103 |
+
"audio_path": str | list[str], # 音频位置
|
| 104 |
+
"question": str, # 针对音频的问题或指令,例如"请将音频内容转写为文字",如果你不需要此字段,请设为空
|
| 105 |
+
"audio_content": str, # 音频��文本形式
|
| 106 |
+
"answer": str, # ground truth答案,如果你不需要此字段(如Open-QA),请设为空
|
| 107 |
+
"subset": str, # 子数据,有时候一个数据集可以被切分为若干个子集,这些子集将被分别评估,独立汇报结果。如果你没有子数据集,填数据集名字即可
|
| 108 |
+
}
|
| 109 |
+
```
|
| 110 |
+
[下载数据集](./data/download_benchmark.py)表明了我们如何下载&处理数据,你可以拿来参考。
|
| 111 |
+
|
| 112 |
+
完成此文件后,你可以将你的数据集添加到适当的类别下,一般而言,继承此类别的父类并填写一些字段后,你的数据集就可用了。例如:
|
| 113 |
+
```
|
| 114 |
+
class Vocalsound(AudioMQADataset):
|
| 115 |
+
DATASET_NAME = 'VocalSound'
|
| 116 |
+
DATASET_SERIES = 'VocalSound'
|
| 117 |
+
AUDIO_TYPE = 'AudioEvent'
|
| 118 |
+
```
|
| 119 |
+
这表明,数据集vocalsound是一个MQA数据集(单选题),它所属的数据集系列是vocalsound,它的AUDIO_TYPE标记为"AudioEvent",说明此数据集是与声音事件(非语音)有关的数据集,这将会在评测时影响一些模型的评测行为。
|
| 120 |
+
|
| 121 |
+
如果你将其下载到数据集缓存目录下,现在你就可以在任意模型上评测此数据集了
|
| 122 |
+
```bash
|
| 123 |
+
bash run_audio.sh --model Kimi-Audio --data vocalsound
|
| 124 |
+
```
|
| 125 |
+
如果你将此文件保存在了其他位置,请在config.yaml中告诉我们
|
| 126 |
+
```yaml
|
| 127 |
+
DATASETS:
|
| 128 |
+
dataset_root: "/path/to/your/dataset/root"
|
| 129 |
+
datasets:
|
| 130 |
+
#example:
|
| 131 |
+
example: "/path/to/your/dataset/example.jsonl"
|
| 132 |
+
Vocalsound: "/path/to/your/dataset/VocalSound.jsonl"
|
| 133 |
+
```
|
| 134 |
+
|
| 135 |
+
## 添加模型
|
| 136 |
+
|
| 137 |
+
在ALMEvalKit评测你的模型也十分容易,你只需要实现generate_inner方法即可,此方法的签名是:
|
| 138 |
+
|
| 139 |
+
```
|
| 140 |
+
def generate_inner(self, msg:dict) -> (str, str)
|
| 141 |
+
```
|
| 142 |
+
**msg** 就是从数据集中的一条数据,它的格式是:
|
| 143 |
+
```python
|
| 144 |
+
{
|
| 145 |
+
"index": int, # 即数据集中一条数据的index,见上
|
| 146 |
+
"audio": list[str] # 大部分情况下长度是1,取audio[0]即可获得此条数据的音频
|
| 147 |
+
"text": str # 即数据的"question"字段,可能为空
|
| 148 |
+
"meta": dict # 数据集的meta信息,如audio_type, name,task等存在这里,如果数据集的一条数据有meta字段,也将会被吸入此字段中
|
| 149 |
+
}
|
| 150 |
+
```
|
| 151 |
+
|
| 152 |
+
此函数的返回是 prompt:str, result:str,prompt为实际送入模型推理的文本,result为模型推理结果。
|
| 153 |
+
|
| 154 |
+
**注意** "实际送入模型推理的文本"不一定等于`msg['text']`,因为我们可以在运行时设定规则篡改它,通常我们会实现一个`get_prompt(msg) -> text`来做这件事。
|
| 155 |
+
|
| 156 |
+
**添加一个模型的最好方式就是copy一个已经实现的模型照猫画虎**
|
| 157 |
+
|
| 158 |
+
## Call for contribution
|
| 159 |
+
|
| 160 |
+
我们希望社区在如下方面共建一个公平、高效、统一的音频大模型评测框架
|
| 161 |
+
|
| 162 |
+
- 增加功能,修改bug,提高代码质量和易用性
|
| 163 |
+
- 支持更多模型和数据集
|
| 164 |
+
- 提高可读性,贡献examples和docs
|
| 165 |
+
|
| 166 |
+
受限于我们所掌握的信息,我们无法为每个模型找到不同任务/数据集下的最佳prompt,我们也欢迎社区提供最佳实践,使得我们的leaderboard能够更加真实的反应模型的极限能力。
|
| 167 |
+
|
| 168 |
+
我们推荐使用[pre-commit](https://pre-commit.com/)来自动格式化你的代码,使你的代码规范与项目保持一致。
|
config.yaml
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
DATASETS:
|
| 2 |
+
dataset_root: "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets"
|
| 3 |
+
datasets:
|
| 4 |
+
#example:
|
| 5 |
+
example: "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets/LibriSpeech/LibriSpeech.jsonl"
|
| 6 |
+
|
| 7 |
+
|
constraints.txt
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
transformers==4.49.0
|
| 2 |
+
tokenizers==0.21.4
|
draw_wer.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import json
|
| 2 |
+
from json import JSONDecodeError
|
| 3 |
+
import matplotlib.pyplot as plt
|
| 4 |
+
|
| 5 |
+
jsonl_path = "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/lora_-5_to_10__7B/Qwen2.5-Omni-7B/kimi_10k_noise_-5_to_10_linear_val5_abs/Qwen2.5-Omni-7B_kimi_10k_noise_-5_to_10_linear_val5_abs_wer_details.jsonl" # 改成你的文件
|
| 6 |
+
out_png = "/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/pic/-5_to_10_val5.png" # 保存路径
|
| 7 |
+
|
| 8 |
+
dec = json.JSONDecoder()
|
| 9 |
+
|
| 10 |
+
def iter_json_from_line(line: str):
|
| 11 |
+
line = line.strip()
|
| 12 |
+
if not line:
|
| 13 |
+
return
|
| 14 |
+
try:
|
| 15 |
+
yield json.loads(line)
|
| 16 |
+
return
|
| 17 |
+
except JSONDecodeError as e:
|
| 18 |
+
# 典型:一行里粘了多个 JSON(Extra data)
|
| 19 |
+
i, n = 0, len(line)
|
| 20 |
+
while i < n:
|
| 21 |
+
while i < n and line[i].isspace():
|
| 22 |
+
i += 1
|
| 23 |
+
if i >= n:
|
| 24 |
+
break
|
| 25 |
+
obj, j = dec.raw_decode(line, i)
|
| 26 |
+
yield obj
|
| 27 |
+
i = j
|
| 28 |
+
|
| 29 |
+
xs, ys = [], []
|
| 30 |
+
cnt_obj = 0
|
| 31 |
+
|
| 32 |
+
with open(jsonl_path, "r", encoding="utf-8") as f:
|
| 33 |
+
for ln, line in enumerate(f, 1):
|
| 34 |
+
try:
|
| 35 |
+
for obj in iter_json_from_line(line):
|
| 36 |
+
cnt_obj += 1
|
| 37 |
+
x = obj.get("index", cnt_obj)
|
| 38 |
+
|
| 39 |
+
y = obj.get("utt_wer", None)
|
| 40 |
+
if y is None:
|
| 41 |
+
y = (obj.get("wer_details") or {}).get("utt_wer", None)
|
| 42 |
+
if y is None:
|
| 43 |
+
continue
|
| 44 |
+
|
| 45 |
+
y = float(y)
|
| 46 |
+
# 兼容:0~1 or 0~100
|
| 47 |
+
if y <= 1.0:
|
| 48 |
+
y *= 100.0
|
| 49 |
+
|
| 50 |
+
xs.append(x)
|
| 51 |
+
ys.append(y)
|
| 52 |
+
except Exception as e:
|
| 53 |
+
# 定位到底是哪一行坏了
|
| 54 |
+
print(f"[ERROR] line {ln} parse failed: {repr(e)}")
|
| 55 |
+
print("line snippet:", repr(line[:200]))
|
| 56 |
+
raise
|
| 57 |
+
|
| 58 |
+
print(f"parsed {cnt_obj} json objects, plotted {len(xs)} points")
|
| 59 |
+
|
| 60 |
+
plt.figure()
|
| 61 |
+
plt.scatter(xs, ys, s=8)
|
| 62 |
+
plt.xlabel("index")
|
| 63 |
+
plt.ylabel("utt_wer (%)")
|
| 64 |
+
plt.ylim(0, 120)
|
| 65 |
+
plt.title("Per-utterance WER scatter")
|
| 66 |
+
plt.savefig(out_png, dpi=300, bbox_inches="tight")
|
| 67 |
+
plt.close()
|
| 68 |
+
|
| 69 |
+
print("saved to:", out_png)
|
ignore-words.txt
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
rouge
|
| 2 |
+
Rouge
|
log_kimilibri.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
log_mini_-5_to_10_all_new_noise.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
log_mini_ea_-5_to_10_all_new_noise.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
log_mini_ea_-5_to_10_new_noise.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
log_mini_origin1.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_default_performance.json
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"task": "ASR",
|
| 3 |
+
"dataset": "voices_dev_clo",
|
| 4 |
+
"model": "Step-Audio-2-mini-all-lora3",
|
| 5 |
+
"date": "2026-01-04 15:40:25.671810",
|
| 6 |
+
"performance": {
|
| 7 |
+
"babb": {
|
| 8 |
+
"wer": 1.76,
|
| 9 |
+
"total": 364
|
| 10 |
+
},
|
| 11 |
+
"musi": {
|
| 12 |
+
"wer": 1.51,
|
| 13 |
+
"total": 371
|
| 14 |
+
},
|
| 15 |
+
"none": {
|
| 16 |
+
"wer": 1.48,
|
| 17 |
+
"total": 380
|
| 18 |
+
},
|
| 19 |
+
"tele": {
|
| 20 |
+
"wer": 1.43,
|
| 21 |
+
"total": 351
|
| 22 |
+
}
|
| 23 |
+
},
|
| 24 |
+
"eval_method": "qwen2-audio-impl"
|
| 25 |
+
}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo_wer_details.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank0.log
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:52:12 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:52:12 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0112/Lab41-SRI-VOiCES-rm1-babb-sp0112-ch123215-sg0025-mc01-stu-clo-dg080.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
| 3 |
+
2026-01-04 12:58:29 | INFO | waiting for other ranks to finish, time elapsed: 10s
|
| 4 |
+
2026-01-04 12:58:39 | INFO | waiting for other ranks to finish, time elapsed: 20s
|
| 5 |
+
2026-01-04 12:58:49 | INFO | waiting for other ranks to finish, time elapsed: 30s
|
| 6 |
+
2026-01-04 12:58:59 | INFO | waiting for other ranks to finish, time elapsed: 40s
|
| 7 |
+
2026-01-04 12:58:59 | INFO | model Step-Audio-2-mini-all-lora3, data voices_dev_clo, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/Step-Audio-2-mini-all-lora3_voices_dev_clo.jsonl.
|
| 8 |
+
2026-01-04 12:58:59 | INFO | skip eval for voices_dev_clo
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank1.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:50:25 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:50:25 | INFO | Msg example: {'index': 2, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0122/Lab41-SRI-VOiCES-rm1-babb-sp0122-ch121729-sg0002-mc02-lav-clo-dg060.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank2.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:52:11 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:52:11 | INFO | Msg example: {'index': 3, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0122/Lab41-SRI-VOiCES-rm1-babb-sp0122-ch121730-sg0014-mc01-stu-clo-dg000.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank3.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:51:24 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:51:24 | INFO | Msg example: {'index': 4, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0159/Lab41-SRI-VOiCES-rm1-babb-sp0159-ch135897-sg0052-mc01-stu-clo-dg100.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank4.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:51:06 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:51:06 | INFO | Msg example: {'index': 5, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0174/Lab41-SRI-VOiCES-rm1-babb-sp0174-ch084280-sg0013-mc02-lav-clo-dg010.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank5.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:51:33 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:51:33 | INFO | Msg example: {'index': 6, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0188/Lab41-SRI-VOiCES-rm1-babb-sp0188-ch135249-sg0029-mc01-stu-clo-dg170.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank6.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:50:37 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:50:37 | INFO | Msg example: {'index': 7, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0205/Lab41-SRI-VOiCES-rm1-babb-sp0205-ch159056-sg0032-mc01-stu-clo-dg020.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_clo/logs/rank7.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:50:40 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_clo
|
| 2 |
+
2026-01-04 12:50:40 | INFO | Msg example: {'index': 8, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_Box_unzip/Development_Data/Automatic_Speech_Recognition/ASR_dev.v2/rm1/babb/sp_0032-1182/sp0208/Lab41-SRI-VOiCES-rm1-babb-sp0208-ch126851-sg0011-mc02-lav-clo-dg070.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices', 'dataset_name': 'voices_dev_clo', 'lang': 'en', 'subset': 'babb'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/Step-Audio-2-mini-all-lora3_voices_dev_test_wer_details.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank0.log
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:24:04 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:24:04 | INFO | Msg example: {'index': 0, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032639-sg0029-mc01-stu-clo-dg160.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
|
| 3 |
+
2026-01-04 12:50:32 | INFO | waiting for other ranks to finish, time elapsed: 10s
|
| 4 |
+
2026-01-04 12:50:42 | INFO | waiting for other ranks to finish, time elapsed: 20s
|
| 5 |
+
2026-01-04 12:50:52 | INFO | waiting for other ranks to finish, time elapsed: 30s
|
| 6 |
+
2026-01-04 12:51:02 | INFO | waiting for other ranks to finish, time elapsed: 40s
|
| 7 |
+
2026-01-04 12:51:12 | INFO | waiting for other ranks to finish, time elapsed: 50s
|
| 8 |
+
2026-01-04 12:51:22 | INFO | waiting for other ranks to finish, time elapsed: 60s
|
| 9 |
+
2026-01-04 12:51:32 | INFO | waiting for other ranks to finish, time elapsed: 70s
|
| 10 |
+
2026-01-04 12:51:42 | INFO | waiting for other ranks to finish, time elapsed: 80s
|
| 11 |
+
2026-01-04 12:51:52 | INFO | waiting for other ranks to finish, time elapsed: 90s
|
| 12 |
+
2026-01-04 12:52:02 | INFO | waiting for other ranks to finish, time elapsed: 100s
|
| 13 |
+
2026-01-04 12:52:12 | INFO | waiting for other ranks to finish, time elapsed: 110s
|
| 14 |
+
2026-01-04 12:52:12 | INFO | model Step-Audio-2-mini-all-lora3, data voices_dev_test, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/Step-Audio-2-mini-all-lora3_voices_dev_test.jsonl.
|
| 15 |
+
2026-01-04 12:52:12 | INFO | skip eval for voices_dev_test
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank1.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:23:40 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:23:40 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032639-sg0029-mc05-stu-far-dg160.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-far'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank2.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:24:00 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:24:00 | INFO | Msg example: {'index': 2, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032658-sg0012-mc05-stu-far-dg070.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-far'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank3.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:23:47 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:23:47 | INFO | Msg example: {'index': 3, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp4899/Lab41-SRI-VOiCES-rm4-babb-sp4899-ch032658-sg0012-mc01-stu-clo-dg070.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank4.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:23:49 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:23:49 | INFO | Msg example: {'index': 4, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp1447/Lab41-SRI-VOiCES-rm4-babb-sp1447-ch130550-sg0026-mc01-stu-clo-dg010.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank5.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:23:46 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:23:46 | INFO | Msg example: {'index': 5, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp1447/Lab41-SRI-VOiCES-rm4-babb-sp1447-ch130550-sg0026-mc05-stu-far-dg010.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-far'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora3/voices_dev_test/logs/rank6.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 12:23:39 | INFO | Running Step-Audio-2-mini-all-lora3 on dataset: voices_dev_test
|
| 2 |
+
2026-01-04 12:23:39 | INFO | Msg example: {'index': 6, 'audio': ['/workspace/intern/pangkaiyu/dg/VOiCES_devkit/distant-16k/speech/test/rm4/babb/sp1447/Lab41-SRI-VOiCES-rm4-babb-sp1447-ch130551-sg0027-mc01-stu-clo-dg140.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'voices_dev', 'dataset_name': 'voices_dev_test', 'lang': 'en', 'subset': 'rm4-babb-clo'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_default_performance.json
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"task": "ASR",
|
| 3 |
+
"dataset": "chime4_dev-real_kimi",
|
| 4 |
+
"model": "Step-Audio-2-mini-all-lora4",
|
| 5 |
+
"date": "2026-01-04 15:40:41.147211",
|
| 6 |
+
"performance": {
|
| 7 |
+
"bus": {
|
| 8 |
+
"wer": 4.41,
|
| 9 |
+
"total": 410
|
| 10 |
+
},
|
| 11 |
+
"caf": {
|
| 12 |
+
"wer": 4.05,
|
| 13 |
+
"total": 410
|
| 14 |
+
},
|
| 15 |
+
"ped": {
|
| 16 |
+
"wer": 3.76,
|
| 17 |
+
"total": 410
|
| 18 |
+
},
|
| 19 |
+
"str": {
|
| 20 |
+
"wer": 4.0,
|
| 21 |
+
"total": 410
|
| 22 |
+
}
|
| 23 |
+
},
|
| 24 |
+
"eval_method": "qwen2-audio-impl"
|
| 25 |
+
}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi_wer_details.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank0.log
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:58 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:58 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C0103_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
| 3 |
+
2026-01-04 13:04:56 | INFO | waiting for other ranks to finish, time elapsed: 10s
|
| 4 |
+
2026-01-04 13:05:06 | INFO | waiting for other ranks to finish, time elapsed: 20s
|
| 5 |
+
2026-01-04 13:05:06 | INFO | model Step-Audio-2-mini-all-lora4, data chime4_dev-real_kimi, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/Step-Audio-2-mini-all-lora4_chime4_dev-real_kimi.jsonl.
|
| 6 |
+
2026-01-04 13:05:06 | INFO | skip eval for chime4_dev-real_kimi
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank1.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:47 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:47 | INFO | Msg example: {'index': 2, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C0105_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank2.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:51 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:51 | INFO | Msg example: {'index': 3, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010C_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank3.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:49 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:49 | INFO | Msg example: {'index': 4, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010G_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank4.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:49 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:49 | INFO | Msg example: {'index': 5, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010J_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank5.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:48 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:48 | INFO | Msg example: {'index': 6, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010K_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank6.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:50 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:50 | INFO | Msg example: {'index': 7, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010L_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/chime4_dev-real_kimi/logs/rank7.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:01:51 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: chime4_dev-real_kimi
|
| 2 |
+
2026-01-04 13:01:51 | INFO | Msg example: {'index': 8, 'audio': ['/workspace/intern/pangkaiyu/dg/CHIME-4/CHiME4/data/audio/16kHz/isolated_1ch_track/dt05_bus_real/F01_050C010O_BUS.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'chime4-dev-real', 'dataset_name': 'chime4_dev-real_kimi', 'lang': 'en', 'subset': 'bus'}}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_default_performance.json
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"task": "ASR",
|
| 3 |
+
"dataset": "noizeus",
|
| 4 |
+
"model": "Step-Audio-2-mini-all-lora4",
|
| 5 |
+
"date": "2026-01-04 15:40:40.185443",
|
| 6 |
+
"performance": {
|
| 7 |
+
"airport_0dB": {
|
| 8 |
+
"wer": 25.21,
|
| 9 |
+
"total": 30
|
| 10 |
+
},
|
| 11 |
+
"airport_10dB": {
|
| 12 |
+
"wer": 2.07,
|
| 13 |
+
"total": 30
|
| 14 |
+
},
|
| 15 |
+
"airport_15dB": {
|
| 16 |
+
"wer": 1.65,
|
| 17 |
+
"total": 30
|
| 18 |
+
},
|
| 19 |
+
"airport_5dB": {
|
| 20 |
+
"wer": 9.92,
|
| 21 |
+
"total": 30
|
| 22 |
+
},
|
| 23 |
+
"babble_0dB": {
|
| 24 |
+
"wer": 38.43,
|
| 25 |
+
"total": 30
|
| 26 |
+
},
|
| 27 |
+
"babble_10dB": {
|
| 28 |
+
"wer": 3.31,
|
| 29 |
+
"total": 30
|
| 30 |
+
},
|
| 31 |
+
"babble_15dB": {
|
| 32 |
+
"wer": 2.07,
|
| 33 |
+
"total": 30
|
| 34 |
+
},
|
| 35 |
+
"babble_5dB": {
|
| 36 |
+
"wer": 4.96,
|
| 37 |
+
"total": 30
|
| 38 |
+
},
|
| 39 |
+
"car_0dB": {
|
| 40 |
+
"wer": 35.54,
|
| 41 |
+
"total": 30
|
| 42 |
+
},
|
| 43 |
+
"car_10dB": {
|
| 44 |
+
"wer": 5.79,
|
| 45 |
+
"total": 30
|
| 46 |
+
},
|
| 47 |
+
"car_15dB": {
|
| 48 |
+
"wer": 2.48,
|
| 49 |
+
"total": 30
|
| 50 |
+
},
|
| 51 |
+
"car_5dB": {
|
| 52 |
+
"wer": 7.85,
|
| 53 |
+
"total": 30
|
| 54 |
+
},
|
| 55 |
+
"exhibition_0dB": {
|
| 56 |
+
"wer": 31.4,
|
| 57 |
+
"total": 30
|
| 58 |
+
},
|
| 59 |
+
"exhibition_10dB": {
|
| 60 |
+
"wer": 4.55,
|
| 61 |
+
"total": 30
|
| 62 |
+
},
|
| 63 |
+
"exhibition_15dB": {
|
| 64 |
+
"wer": 2.48,
|
| 65 |
+
"total": 30
|
| 66 |
+
},
|
| 67 |
+
"exhibition_5dB": {
|
| 68 |
+
"wer": 7.44,
|
| 69 |
+
"total": 30
|
| 70 |
+
},
|
| 71 |
+
"restaurant_0dB": {
|
| 72 |
+
"wer": 30.17,
|
| 73 |
+
"total": 30
|
| 74 |
+
},
|
| 75 |
+
"restaurant_10dB": {
|
| 76 |
+
"wer": 1.65,
|
| 77 |
+
"total": 30
|
| 78 |
+
},
|
| 79 |
+
"restaurant_15dB": {
|
| 80 |
+
"wer": 2.89,
|
| 81 |
+
"total": 30
|
| 82 |
+
},
|
| 83 |
+
"restaurant_5dB": {
|
| 84 |
+
"wer": 6.61,
|
| 85 |
+
"total": 30
|
| 86 |
+
},
|
| 87 |
+
"station_0dB": {
|
| 88 |
+
"wer": 23.97,
|
| 89 |
+
"total": 30
|
| 90 |
+
},
|
| 91 |
+
"station_10dB": {
|
| 92 |
+
"wer": 3.31,
|
| 93 |
+
"total": 30
|
| 94 |
+
},
|
| 95 |
+
"station_15dB": {
|
| 96 |
+
"wer": 2.07,
|
| 97 |
+
"total": 30
|
| 98 |
+
},
|
| 99 |
+
"station_5dB": {
|
| 100 |
+
"wer": 9.92,
|
| 101 |
+
"total": 30
|
| 102 |
+
},
|
| 103 |
+
"street_0dB": {
|
| 104 |
+
"wer": 39.26,
|
| 105 |
+
"total": 30
|
| 106 |
+
},
|
| 107 |
+
"street_10dB": {
|
| 108 |
+
"wer": 5.37,
|
| 109 |
+
"total": 30
|
| 110 |
+
},
|
| 111 |
+
"street_15dB": {
|
| 112 |
+
"wer": 1.65,
|
| 113 |
+
"total": 30
|
| 114 |
+
},
|
| 115 |
+
"street_5dB": {
|
| 116 |
+
"wer": 13.64,
|
| 117 |
+
"total": 30
|
| 118 |
+
}
|
| 119 |
+
},
|
| 120 |
+
"eval_method": "qwen2-audio-impl"
|
| 121 |
+
}
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus_wer_details.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank0.log
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:00:54 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: noizeus
|
| 2 |
+
2026-01-04 13:00:54 | INFO | Msg example: {'index': 0, 'audio': ['/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets/noizeus/noizeus/airport/0dB/sp01_airport_sn0.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'noizeus', 'dataset_name': 'noizeus', 'lang': 'en', 'subset': 'airport_0dB'}}
|
| 3 |
+
2026-01-04 13:01:58 | INFO | waiting for other ranks to finish, time elapsed: 10s
|
| 4 |
+
2026-01-04 13:01:58 | INFO | model Step-Audio-2-mini-all-lora4, data noizeus, all 8 result merged to mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/Step-Audio-2-mini-all-lora4_noizeus.jsonl.
|
| 5 |
+
2026-01-04 13:01:58 | INFO | skip eval for noizeus
|
mini_whole/mini-encoder+align-whole1226_signal_new1-low-lora-1gpu-bs16_1_gckF2000/Step-Audio-2-mini-all-lora4/noizeus/logs/rank1.log
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-01-04 13:00:54 | INFO | Running Step-Audio-2-mini-all-lora4 on dataset: noizeus
|
| 2 |
+
2026-01-04 13:00:54 | INFO | Msg example: {'index': 1, 'audio': ['/workspace/intern/pangkaiyu/Kimi-Audio/Kimi-Audio-Evalkit/data/downloaded_datasets/noizeus/noizeus/airport/0dB/sp02_airport_sn0.wav'], 'text': 'Please transcribe the spoken content into written text.', 'meta': {'task': 'ASR', 'interactive': 'Audio-analysis', 'audio_type': 'Speech', 'dataset_series': 'noizeus', 'dataset_name': 'noizeus', 'lang': 'en', 'subset': 'airport_0dB'}}
|