| { |
| "model_type": "audio-jepa", |
| "architecture": "VisionTransformer", |
| "citation": "Tuncay, Labbé, Benetos, Pellegrini. Audio-JEPA: Joint-Embedding Predictive Architecture for Audio Representation Learning. ICME 2025 (arXiv 2507.02915)", |
| "audio": { |
| "sample_rate": 32000, |
| "clip_length_s": 10, |
| "n_mels": 128, |
| "target_time_bins": 256, |
| "mel_spec_backend": "torchaudio.compliance.kaldi.fbank", |
| "f_min": 20, |
| "f_max": null, |
| "log_fbank": true, |
| "window": "hanning" |
| }, |
| "patch_embed": { |
| "patch_size": [16, 16], |
| "in_chans": 1, |
| "num_patches": 128, |
| "grid_time": 8, |
| "grid_freq": 16 |
| }, |
| "encoder": { |
| "embed_dim": 768, |
| "depth": 12, |
| "num_heads": 12, |
| "mlp_ratio": 4.0, |
| "qkv_bias": true, |
| "use_flash_attn": true, |
| "norm_layer": "LayerNorm", |
| "cls_token": false, |
| "pos_embed": "2d_sincos" |
| }, |
| "output": { |
| "shape_per_10s_clip": [128, 768], |
| "temporal_positions_per_second": 1.6, |
| "tokens_per_second": 12.8, |
| "note": "Each patch spans 16 mel bins x 16 time bins ≈ 625 ms. The 12.8 Hz rate counts 8 temporal x 16 frequency patches per second; effective temporal resolution is 1.6 positions/s." |
| }, |
| "training": { |
| "framework": "PyTorch Lightning + Hydra", |
| "pretraining_corpus": "AudioSet unlabeled (~5.3k h after silence filtering)", |
| "target_module": "target_encoder (EMA of encoder)", |
| "predictor": { |
| "type": "VisionTransformerPredictor", |
| "depth": 6, |
| "num_heads": 12, |
| "mlp_ratio": 4.0 |
| }, |
| "loss": "norm_mse (with norm_pix_loss=True)", |
| "note": "Only the encoder is used downstream. Predictor + target encoder are discarded at inference." |
| }, |
| "known_speech_gap": { |
| "description": "This model is designed for generic audio representation. The 16x16 patch shape reflects a compromise across speech, music, and environmental sounds. Speech-only downstream performance is lower than speech-specific SSL models (wav2vec2, HuBERT, Whisper). See the paper's X-ARES tables and the follow-up BEST-RQ-2 (arXiv 2606.30700) for a quantitative comparison.", |
| "suggested_alternative_for_speech": "For speech-heavy downstream tasks, consider BEST-RQ-2 (same author, arXiv 2606.30700, MIT weights at ltuncay/BEST-RQ-2) which shares the encoder architecture, or wait for the upcoming BEST-RQ-2.1 with explicit speech-oriented improvements." |
| } |
| } |
|
|