Update custom model files, README, and requirements

Browse files

Files changed (5) hide show

.gitattributes +2 -35
README.md +247 -57
asr_config.py +13 -22
handler.py +73 -0
requirements.txt +5 -0

.gitattributes CHANGED Viewed

@@ -1,36 +1,3 @@
-*.7z filter=lfs diff=lfs merge=lfs -text
-*.arrow filter=lfs diff=lfs merge=lfs -text
-*.bin filter=lfs diff=lfs merge=lfs -text
-*.bz2 filter=lfs diff=lfs merge=lfs -text
-*.ckpt filter=lfs diff=lfs merge=lfs -text
-*.ftz filter=lfs diff=lfs merge=lfs -text
-*.gz filter=lfs diff=lfs merge=lfs -text
-*.h5 filter=lfs diff=lfs merge=lfs -text
-*.joblib filter=lfs diff=lfs merge=lfs -text
-*.lfs.* filter=lfs diff=lfs merge=lfs -text
-*.mlmodel filter=lfs diff=lfs merge=lfs -text
-*.model filter=lfs diff=lfs merge=lfs -text
-*.msgpack filter=lfs diff=lfs merge=lfs -text
-*.npy filter=lfs diff=lfs merge=lfs -text
-*.npz filter=lfs diff=lfs merge=lfs -text
-*.onnx filter=lfs diff=lfs merge=lfs -text
-*.ot filter=lfs diff=lfs merge=lfs -text
-*.parquet filter=lfs diff=lfs merge=lfs -text
-*.pb filter=lfs diff=lfs merge=lfs -text
-*.pickle filter=lfs diff=lfs merge=lfs -text
-*.pkl filter=lfs diff=lfs merge=lfs -text
-*.pt filter=lfs diff=lfs merge=lfs -text
-*.pth filter=lfs diff=lfs merge=lfs -text
-*.rar filter=lfs diff=lfs merge=lfs -text
 *.safetensors filter=lfs diff=lfs merge=lfs -text
-saved_model/**/* filter=lfs diff=lfs merge=lfs -text
-*.tar.* filter=lfs diff=lfs merge=lfs -text
-*.tar filter=lfs diff=lfs merge=lfs -text
-*.tflite filter=lfs diff=lfs merge=lfs -text
-*.tgz filter=lfs diff=lfs merge=lfs -text
-*.wasm filter=lfs diff=lfs merge=lfs -text
-*.xz filter=lfs diff=lfs merge=lfs -text
-*.zip filter=lfs diff=lfs merge=lfs -text
-*.zst filter=lfs diff=lfs merge=lfs -text
-*tfevents* filter=lfs diff=lfs merge=lfs -text
-tokenizer.json filter=lfs diff=lfs merge=lfs -text

 *.safetensors filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+tokenizer_config.json -filter -diff -merge text

README.md CHANGED Viewed

@@ -1,77 +1,267 @@
 ---
-library_name: transformers
 tags:
-- generated_from_trainer
-model-index:
-- name: tiny-audio-omni
-  results: []
 ---
-<!-- This model card has been generated automatically according to the information the Trainer had access to. You
-should probably proofread and complete it, then remove this comment. -->
-# tiny-audio-omni
-This model is a fine-tuned version of [](https://huggingface.co/) on the None dataset.
-It achieves the following results on the evaluation set:
-- Loss: 0.4575
-## Model description
-More information needed
-## Intended uses & limitations
-More information needed
-## Training and evaluation data
-More information needed
-## Training procedure
-### Training hyperparameters
-The following hyperparameters were used during training:
-- learning_rate: 0.001
-- train_batch_size: 14
-- eval_batch_size: 14
-- seed: 42
-- gradient_accumulation_steps: 4
-- total_train_batch_size: 56
-- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
-- lr_scheduler_type: polynomial
-- lr_scheduler_warmup_steps: 1000
-- num_epochs: 1
-- label_smoothing_factor: 0.1
-### Training results
-| Training Loss | Epoch  | Step  | Validation Loss |
-|:-------------:|:------:|:-----:|:---------------:|
-| 2.1239        | 0.0534 | 1000  | 0.4662          |
-| 2.0815        | 0.1069 | 2000  | 0.4654          |
-| 2.0997        | 0.1603 | 3000  | 0.4644          |
-| 2.0654        | 0.2137 | 4000  | 0.4634          |
-| 2.0897        | 0.2672 | 5000  | 0.4625          |
-| 2.0523        | 0.3206 | 6000  | 0.4618          |
-| 2.0583        | 0.3740 | 7000  | 0.4616          |
-| 2.0573        | 0.4274 | 8000  | 0.4608          |
-| 2.0345        | 0.4809 | 9000  | 0.4603          |
-| 2.0328        | 0.5343 | 10000 | 0.4598          |
-| 2.0610        | 0.5877 | 11000 | 0.4593          |
-| 2.0336        | 0.6412 | 12000 | 0.4592          |
-| 2.0445        | 0.6946 | 13000 | 0.4588          |
-| 2.0572        | 0.7480 | 14000 | 0.4582          |
-| 2.0349        | 0.8015 | 15000 | 0.4582          |
-| 2.0164        | 0.8549 | 16000 | 0.4579          |
-| 2.0246        | 0.9083 | 17000 | 0.4576          |
-| 2.0219        | 0.9617 | 18000 | 0.4575          |
-### Framework versions
-- Transformers 5.0.0
-- Pytorch 2.8.0+cu128
-- Datasets 3.6.0
-- Tokenizers 0.22.2

 ---
+license: mit
+language:
+- en
+datasets:
+- speechbrain/LoquaciousSet
+base_model:
+- zai-org/GLM-ASR-Nano-2512
+- Qwen/Qwen3-0.6B
+pipeline_tag: automatic-speech-recognition
 tags:
+- asr
+- speech-recognition
+- audio
+- qwen
+- glm-asr
+library_name: transformers
 ---
+# Tiny Audio
+A speech recognition model trained in 24 hours on a single GPU for ~$12. Built with [Tiny Audio](https://github.com/alexkroman/tiny-audio)—a minimal, hackable ASR framework.
+## Quick Start
+```python
+from transformers import pipeline
+pipe = pipeline("automatic-speech-recognition", model="mazesmazes/tiny-audio", trust_remote_code=True)
+result = pipe("audio.wav")
+print(result["text"])
+```
+## Usage Examples
+### Basic Transcription
+```python
+from transformers import pipeline
+pipe = pipeline("automatic-speech-recognition", model="mazesmazes/tiny-audio", trust_remote_code=True)
+# From file
+result = pipe("audio.wav")
+print(result["text"])
+# From URL
+result = pipe("https://example.com/audio.mp3")
+# From numpy array (must be 16kHz)
+import numpy as np
+audio = np.random.randn(16000).astype(np.float32)  # 1 second
+result = pipe(audio)
+```
+### Batch Processing
+```python
+# Process multiple files
+files = ["audio1.wav", "audio2.wav", "audio3.wav"]
+results = pipe(files, batch_size=4)
+for r in results:
+    print(r["text"])
+```
+### Word-Level Timestamps
+```python
+result = pipe("audio.wav", return_timestamps="word")
+# Returns:
+# {
+#   "text": "hello world",
+#   "chunks": [
+#     {"text": "hello", "timestamp": (0.0, 0.5)},
+#     {"text": "world", "timestamp": (0.6, 1.0)}
+#   ]
+# }
+```
+### Streaming Inference
+```python
+from tiny_audio import ASRModel, ASRProcessor
+import torch
+model = ASRModel.from_pretrained("mazesmazes/tiny-audio")
+processor = ASRProcessor.from_pretrained("mazesmazes/tiny-audio")
+# Load and process audio
+import librosa
+audio, sr = librosa.load("audio.wav", sr=16000)
+inputs = processor(audio, sampling_rate=16000, return_tensors="pt")
+# Stream tokens
+for token in model.generate_streaming(inputs["input_features"]):
+    print(token, end="", flush=True)
+```
+### Using with torch directly
+```python
+from tiny_audio import ASRModel, ASRProcessor
+import torch
+import librosa
+# Load model and processor
+model = ASRModel.from_pretrained("mazesmazes/tiny-audio")
+processor = ASRProcessor.from_pretrained("mazesmazes/tiny-audio")
+# Load audio (16kHz)
+audio, sr = librosa.load("audio.wav", sr=16000)
+# Process
+inputs = processor(audio, sampling_rate=16000, return_tensors="pt")
+# Generate
+with torch.no_grad():
+    output = model.generate(
+        input_features=inputs["input_features"],
+        attention_mask=inputs["attention_mask"],
+        max_new_tokens=256
+    )
+# Decode
+text = processor.batch_decode(output, skip_special_tokens=True)[0]
+print(text)
+```
+### GPU Inference
+```python
+import torch
+pipe = pipeline(
+    "automatic-speech-recognition",
+    model="mazesmazes/tiny-audio",
+    trust_remote_code=True,
+    device="cuda"  # or device=0
+)
+```
+### Half Precision
+```python
+pipe = pipeline(
+    "automatic-speech-recognition",
+    model="mazesmazes/tiny-audio",
+    trust_remote_code=True,
+    torch_dtype=torch.float16,
+    device="cuda"
+)
+```
+## Architecture
+```
+Audio (16kHz) → GLM-ASR Encoder (frozen) → MLP Projector (trained) → Qwen3 (frozen) → Text
+```
+Only the projector is trained (~12M params). The encoder and decoder remain frozen, leveraging their pretrained knowledge.
+| Component | Model | Parameters | Status |
+|-----------|-------|------------|--------|
+| Audio Encoder | GLM-ASR-Nano-2512 | ~600M | Frozen |
+| Projector | 2-layer MLP | ~12M | Trained |
+| Language Model | Qwen3-0.6B | ~600M | Frozen |
+### How It Works
+1. **Audio Encoder**: GLM-ASR converts 16kHz audio into frame-level embeddings (768-dim)
+2. **Projector**: A 2-layer MLP with frame stacking bridges the audio and text embedding spaces
+3. **Language Model**: Qwen3 generates text autoregressively, conditioned on the projected audio
+The projector reduces sequence length via frame stacking: `output_len = (input_len - 5) // 5 + 1`
+## Model Specifications
+| Specification | Value |
+|---------------|-------|
+| Input | Audio (16kHz mono) |
+| Output | Text transcription |
+| Max Audio Length | ~30 seconds (limited by encoder) |
+| Vocabulary | Qwen3 tokenizer |
+| Languages | English only |
+| Generation | Greedy decoding (num_beams=1, do_sample=False) |
+## Training Details
+| | |
+|---|---|
+| **Dataset** | LoquaciousSet (25,000 hours) |
+| **Hardware** | Single NVIDIA A40 |
+| **Time** | ~24 hours |
+| **Cost** | ~$12 |
+| **Optimizer** | AdamW |
+| **Learning Rate** | 1e-4 |
+| **Batch Size** | 4 |
+| **Steps** | 50,000 |
+## Limitations
+- **English only**: Not trained on other languages
+- **Sample rate**: Expects 16kHz audio (other rates resampled automatically)
+- **Audio length**: Best for clips under 30 seconds
+- **Accuracy**: May degrade on:
+  - Heavily accented speech
+  - Noisy or low-quality audio
+  - Domain-specific terminology
+  - Overlapping speakers
+- **No punctuation**: Output is lowercase without punctuation by default
+## Requirements
+```
+transformers>=4.40.0
+torch>=2.0.0
+torchaudio>=2.0.0
+```
+Optional for streaming:
+```
+librosa
+soundfile
+```
+## Files
+| File | Description |
+|------|-------------|
+| `config.json` | Model configuration |
+| `model.safetensors` | Projector weights (~48MB) |
+| `preprocessor_config.json` | Audio preprocessing config |
+| `tokenizer.json` | Tokenizer |
+| `tokenizer_config.json` | Tokenizer config |
+| `special_tokens_map.json` | Special tokens |
+Note: Only the projector weights are stored. The encoder (GLM-ASR) and decoder (Qwen3) are loaded from their respective HuggingFace repos.
+## Citation
+If you use this model, please cite:
+```bibtex
+@misc{tinyaudio2024,
+  author = {Alex Kroman},
+  title = {Tiny Audio: Minimal ASR Training},
+  year = {2024},
+  publisher = {GitHub},
+  url = {https://github.com/alexkroman/tiny-audio}
+}
+```
+## Links
+- [GitHub Repository](https://github.com/alexkroman/tiny-audio) - Train your own model
+- [Free 3.5-hour Course](https://github.com/alexkroman/tiny-audio/blob/main/docs/course/0-course-overview.md) - Learn ASR from scratch
+- [Live Demo](https://huggingface.co/spaces/mazesmazes/tiny-audio) - Try it in your browser
+## Acknowledgments
+- [GLM-ASR](https://huggingface.co/zai-org/GLM-ASR-Nano-2512) for the audio encoder
+- [Qwen3](https://huggingface.co/Qwen/Qwen3-0.6B) for the language model
+- [LoquaciousSet](https://huggingface.co/datasets/speechbrain/LoquaciousSet) for training data
+## License
+MIT

asr_config.py CHANGED Viewed

@@ -152,28 +152,19 @@ class ASRConfig(transformers.PretrainedConfig):
         ]
         self.freeze_projector = freeze_projector
-        # Generation parameters (use explicit value if provided, else use default)
-        self.num_beams = num_beams if num_beams is not None else generation_defaults["num_beams"]
-        self.max_new_tokens = (
-            max_new_tokens if max_new_tokens is not None else generation_defaults["max_new_tokens"]
-        )
-        self.min_new_tokens = (
-            min_new_tokens if min_new_tokens is not None else generation_defaults["min_new_tokens"]
-        )
-        self.repetition_penalty = (
-            repetition_penalty
-            if repetition_penalty is not None
-            else generation_defaults["repetition_penalty"]
-        )
-        self.length_penalty = (
-            length_penalty if length_penalty is not None else generation_defaults["length_penalty"]
-        )
-        self.no_repeat_ngram_size = (
-            no_repeat_ngram_size
-            if no_repeat_ngram_size is not None
-            else generation_defaults["no_repeat_ngram_size"]
-        )
-        self.use_cache = use_cache if use_cache is not None else generation_defaults["use_cache"]
         self.do_sample = do_sample
         self.enable_thinking = enable_thinking
         self.temperature = temperature

         ]
         self.freeze_projector = freeze_projector
+        # Generation parameters: check named param first, then kwargs (from config.json), then default
+        def get_gen_param(name, named_value):
+            if named_value is not None:
+                return named_value
+            return kwargs.get(name, generation_defaults[name])
+        self.num_beams = get_gen_param("num_beams", num_beams)
+        self.max_new_tokens = get_gen_param("max_new_tokens", max_new_tokens)
+        self.min_new_tokens = get_gen_param("min_new_tokens", min_new_tokens)
+        self.repetition_penalty = get_gen_param("repetition_penalty", repetition_penalty)
+        self.length_penalty = get_gen_param("length_penalty", length_penalty)
+        self.no_repeat_ngram_size = get_gen_param("no_repeat_ngram_size", no_repeat_ngram_size)
+        self.use_cache = get_gen_param("use_cache", use_cache)
         self.do_sample = do_sample
         self.enable_thinking = enable_thinking
         self.temperature = temperature

handler.py ADDED Viewed

	@@ -0,0 +1,73 @@

+"""Custom inference handler for HuggingFace Inference Endpoints."""
+from typing import Any, Dict, List, Union
+try:
+    # For remote execution, imports are relative
+    from .asr_modeling import ASRModel
+    from .asr_pipeline import ASRPipeline
+except ImportError:
+    # For local execution, imports are not relative
+    from asr_modeling import ASRModel  # type: ignore[no-redef]
+    from asr_pipeline import ASRPipeline  # type: ignore[no-redef]
+class EndpointHandler:
+    """HuggingFace Inference Endpoints handler for ASR model.
+    Handles model loading, warmup, and inference requests for deployment
+    on HuggingFace Inference Endpoints or similar services.
+    """
+    def __init__(self, path: str = ""):
+        """Initialize the endpoint handler.
+        Args:
+            path: Path to model directory or HuggingFace model ID
+        """
+        import os
+        import nltk
+        nltk.download("punkt_tab", quiet=True)
+        os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
+        # Prepare model kwargs - let transformers handle device placement
+        model_kwargs = {
+            "device_map": "auto",
+            "torch_dtype": "auto",
+            "low_cpu_mem_usage": True,
+        }
+        # Load model (this loads the model, tokenizer, and feature extractor)
+        self.model = ASRModel.from_pretrained(path, **model_kwargs)
+        # Get device from model for pipeline
+        self.device = next(self.model.parameters()).device
+        # Instantiate custom pipeline - it will get feature_extractor and tokenizer from model
+        self.pipe = ASRPipeline(
+            model=self.model,
+            feature_extractor=self.model.feature_extractor,
+            tokenizer=self.model.tokenizer,
+            device=self.device,
+        )
+    def __call__(self, data: Dict[str, Any]) -> Union[Dict[str, Any], List[Dict[str, Any]]]:
+        """Process an inference request.
+        Args:
+            data: Request data containing 'inputs' (audio path/bytes) and optional 'parameters'
+        Returns:
+            Transcription result with 'text' key
+        """
+        inputs = data.get("inputs")
+        if inputs is None:
+            raise ValueError("Missing 'inputs' in request data")
+        # Pass through any parameters from request, let model config provide defaults
+        params = data.get("parameters", {})
+        return self.pipe(inputs, **params)

requirements.txt ADDED Viewed

	@@ -0,0 +1,5 @@

+# Core dependencies for tiny-audio model inference
+# This file is pushed to HuggingFace for model repository
+# Transformers - main library for model loading and inference
+transformers>=4.57.0