Automatic Speech Recognition
Transformers
Safetensors
English
asr_model
asr
speech-recognition
speech-to-text
audio
speech-llm
word-timestamps
speaker-diarization
qwen
granite-speech
lora
custom_code
Eval Results (legacy)
Instructions to use mazesmazes/tiny-audio with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use mazesmazes/tiny-audio with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("automatic-speech-recognition", model="mazesmazes/tiny-audio", trust_remote_code=True)# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForSpeechSeq2Seq model = AutoModelForSpeechSeq2Seq.from_pretrained("mazesmazes/tiny-audio", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download asr_processing.py from mazesmazes/tiny-audio: direct link, hf CLI and curl.
- Browser
- Download file 16.3 kB
-
https://huggingface.co/mazesmazes/tiny-audio/resolve/main/asr_processing.py
- Command line
-
hf download hf://mazesmazes/tiny-audio/asr_processing.py
-
curl -L -o asr_processing.py https://huggingface.co/mazesmazes/tiny-audio/resolve/main/asr_processing.py
16.3 kB
| """Processor that turns raw audio (and optional text) into model inputs.""" | |
| from collections.abc import Mapping, Sequence | |
| from typing import TYPE_CHECKING, Any, ClassVar, cast, overload | |
| import numpy as np | |
| import numpy.typing as npt | |
| import torch | |
| import transformers | |
| from torch.nn.utils.rnn import pad_sequence | |
| from transformers import ( | |
| BatchFeature, | |
| PreTrainedTokenizerBase, | |
| ProcessorMixin, | |
| SequenceFeatureExtractor, | |
| ) | |
| if TYPE_CHECKING: | |
| from .asr_config import ( | |
| DEFAULT_ENCODER_CONV_LAYERS, | |
| ASRConfig, | |
| ConvLayerSpec, | |
| compute_encoder_output_length, | |
| ) | |
| from .asr_types import AudioFeatureExtractor, AudioInput, PreparedChunk, Waveform | |
| from .projectors import MLPAudioProjector | |
| else: | |
| try: | |
| from .asr_config import ( | |
| DEFAULT_ENCODER_CONV_LAYERS, | |
| ASRConfig, | |
| ConvLayerSpec, | |
| compute_encoder_output_length, | |
| ) | |
| from .asr_types import AudioInput, PreparedChunk | |
| except ImportError: # flat layout on the Hub: sibling modules, no package | |
| from asr_config import ( | |
| DEFAULT_ENCODER_CONV_LAYERS, | |
| ASRConfig, | |
| ConvLayerSpec, | |
| compute_encoder_output_length, | |
| ) | |
| from asr_types import AudioInput, PreparedChunk | |
| def collate_chunks(prepared: Sequence[PreparedChunk]) -> PreparedChunk: | |
| """Pad prepared chunks to the longest and stack them into one batch. | |
| The time axis is whichever feature axis matches the mask's length (Granite's | |
| features are `(1, T, D)`, Whisper's `(1, D, T)`); padded frames are zeros | |
| with a 0 in the mask, which the encoder honours (`encoder_attention_mask`). | |
| """ | |
| longest = max(int(p["attention_mask"].shape[-1]) for p in prepared) | |
| features: list[torch.Tensor] = [] | |
| masks: list[torch.Tensor] = [] | |
| for p in prepared: | |
| feats, mask = p["input_features"], p["attention_mask"] | |
| length = int(mask.shape[-1]) | |
| time_axis = 1 if feats.shape[1] == length else feats.dim() - 1 | |
| pad = longest - length | |
| # F.pad lists (left, right) pairs from the LAST axis backwards. | |
| spec = [0, 0] * (feats.dim() - 1 - time_axis) + [0, pad] | |
| features.append(torch.nn.functional.pad(feats, spec)) | |
| masks.append(torch.nn.functional.pad(mask, (0, pad))) | |
| return {"input_features": torch.cat(features), "attention_mask": torch.cat(masks)} | |
| # The instruction the model trained on (scripts/train_collator.py); the model | |
| # and processor both default to it. | |
| DEFAULT_TRANSCRIBE_PROMPT = "Transcribe the speech to text" | |
| def render_audio_prompt( | |
| tokenizer: PreTrainedTokenizerBase, | |
| audio_token: str, | |
| num_audio_tokens: int, | |
| prompt: str | None, | |
| text: str | None = None, | |
| ) -> torch.Tensor: | |
| """Tokenize one chat prompt carrying exactly `num_audio_tokens` placeholders. | |
| The user turn is the placeholders, then `prompt` (if any); `text`, when | |
| given, is the assistant's reply, otherwise the generation prompt is added. | |
| """ | |
| if num_audio_tokens > 0: | |
| user_content = audio_token * num_audio_tokens | |
| if prompt: | |
| user_content += " " + prompt | |
| else: | |
| user_content = prompt or "" | |
| messages = [{"role": "user", "content": user_content}] | |
| if text is not None: | |
| messages.append({"role": "assistant", "content": text}) | |
| # With `tokenize=True, return_tensors="pt"` the ids come back as tensors. | |
| tokenized = cast( | |
| "torch.Tensor | Mapping[str, torch.Tensor]", | |
| tokenizer.apply_chat_template( | |
| messages, | |
| tokenize=True, | |
| add_generation_prompt=(text is None), | |
| return_tensors="pt", | |
| enable_thinking=False, # Disable Qwen3 thinking mode for ASR | |
| ), | |
| ) | |
| # apply_chat_template returns a bare tensor or a BatchEncoding/mapping. | |
| ids = tokenized if isinstance(tokenized, torch.Tensor) else tokenized["input_ids"] | |
| return (ids[0] if ids.dim() > 1 else ids).to(torch.long) | |
| def left_pad_prompt_rows( | |
| rows: list[torch.Tensor], tokenizer: PreTrainedTokenizerBase | |
| ) -> tuple[torch.Tensor, torch.Tensor]: | |
| """Stack per-sample prompt rows into a left-padded batch: `(input_ids, attention_mask)`. | |
| Left, not right: these feed `generate`, so padding must not sit between | |
| the prompt and the first generated token. Pads with the tokenizer's pad | |
| token, falling back to eos, then 0. Pad positions never carry | |
| `audio_token_id`, so the model's masked_scatter is unaffected. | |
| """ | |
| # transformers types special-token ids as any token value; a single id is an int. | |
| pad_id = cast("int | None", tokenizer.pad_token_id) | |
| if pad_id is None: | |
| pad_id = cast("int | None", tokenizer.eos_token_id) or 0 | |
| input_ids = pad_sequence(rows, batch_first=True, padding_value=int(pad_id), padding_side="left") | |
| # Padded from ones rather than `input_ids != pad_id`: a real token may | |
| # equal `pad_id` when pad falls back to eos. | |
| attention_mask = pad_sequence( | |
| [torch.ones_like(row) for row in rows], batch_first=True, padding_side="left" | |
| ) | |
| return input_ids, attention_mask | |
| def prepend_lead_in[ScalarT: np.generic]( | |
| audio: npt.NDArray[ScalarT], sampling_rate: int, seconds: float | None | |
| ) -> npt.NDArray[ScalarT]: ... | |
| def prepend_lead_in[ScalarT: np.generic]( | |
| audio: list[npt.NDArray[ScalarT]], sampling_rate: int, seconds: float | None | |
| ) -> list[npt.NDArray[ScalarT]]: ... | |
| def prepend_lead_in(audio: AudioInput, sampling_rate: int, seconds: float | None) -> AudioInput: ... | |
| def prepend_lead_in(audio: AudioInput, sampling_rate: int, seconds: float | None) -> AudioInput: | |
| """Prepend `seconds` of silence to a waveform (or each waveform in a list). | |
| Peoples ships fixed ~15s grid cuts rather than sentence-aligned segments, | |
| so a clip routinely opens mid-word and the model declines to emit the | |
| partial first token. Measured on 500 Peoples clips with a paired | |
| bootstrap: 20.51% -> 19.28% WER (delta -1.22, CI [-1.83, -0.64]) and | |
| utterances dropping a leading reference word fall 258/460 -> 170/460. | |
| CommonVoice, whose clips already start cleanly, is unaffected (+0.30, | |
| CI [-0.43, +1.17]). | |
| Inference only. Training feeds raw audio through the collator, so this is | |
| a test-time transform, and it recovers two thirds of the dropped onsets | |
| rather than all of them -- the remainder are clips whose first syllable | |
| was never recorded, which no amount of lead-in reconstructs. | |
| """ | |
| if not seconds or seconds <= 0: | |
| return audio | |
| if isinstance(audio, (list, tuple)) and audio and not isinstance(audio[0], (int, float)): | |
| batch = cast("Sequence[Waveform]", audio) | |
| return [cast("Waveform", prepend_lead_in(a, sampling_rate, seconds)) for a in batch] | |
| waveform = cast("Waveform", audio) | |
| pad = round(sampling_rate * seconds) | |
| if pad <= 0: | |
| return waveform | |
| arr: npt.NDArray[Any] = np.asarray(waveform) | |
| padded: npt.NDArray[Any] = np.pad(arr, (pad, 0)) | |
| return padded | |
| # Audio is transcribed in chunks cut at the quietest point between | |
| # these lengths. The model trained on clips of at most 19 s; 18 leaves room for | |
| # the inference lead-in. Short clips are one chunk, so their text is unchanged. | |
| CHUNK_MAX_S = 18.0 | |
| CHUNK_MIN_S = 8.0 | |
| def chunk_bounds( | |
| audio: npt.NDArray[np.float32], | |
| sample_rate: int, | |
| max_s: float = CHUNK_MAX_S, | |
| min_s: float = CHUNK_MIN_S, | |
| ) -> list[tuple[int, int]]: | |
| """Sample ranges of at most `max_s`, each cut at the quietest 100 ms frame after `min_s`.""" | |
| frame = int(0.1 * sample_rate) | |
| bounds: list[tuple[int, int]] = [] | |
| start, n = 0, len(audio) | |
| while n - start > max_s * sample_rate: | |
| lo = start + int(min_s * sample_rate) | |
| hi = start + int(max_s * sample_rate) | |
| cut = lo + int(np.argmin(_frame_rms(audio[lo:hi], frame))) * frame + frame // 2 | |
| bounds.append((start, cut)) | |
| start = cut | |
| bounds.append((start, n)) | |
| return bounds | |
| def _frame_rms(audio: npt.NDArray[np.float32], frame: int) -> npt.NDArray[np.float32]: | |
| """RMS of each whole `frame`-sample frame of `audio` (a trailing partial frame is dropped).""" | |
| k = len(audio) // frame | |
| return np.sqrt(np.mean(np.square(audio[: k * frame].reshape(k, frame)), axis=1)) | |
| # A chunk whose loudest 100 ms frame is this far below the recording's speech | |
| # level (its 95th-percentile frame) holds no speech, only the room tone after | |
| # the talker stopped. Decoded, such a tail comes back as a memorized sentence | |
| # ("The film was directed by the director of the same name.", 0.7 WER on | |
| # CommonVoice) or a stray "the"/"ok". On the cached eval clips over 18 s every | |
| # noise-only chunk sat at -36 dB or below and every chunk with speech at | |
| # -17.5 dB or above; -30 keeps the wider margin on the speech side. | |
| QUIET_CHUNK_DB = -30.0 | |
| def audible_chunks( | |
| audio: npt.NDArray[np.float32], bounds: list[tuple[int, int]], sample_rate: int | |
| ) -> list[npt.NDArray[np.float32]]: | |
| """The chunks of `audio` at `bounds`, those quieter than `QUIET_CHUNK_DB` emptied. | |
| An empty chunk `is_silent`, so it transcribes as "" without the model. A | |
| one-chunk recording is never emptied: its loudest frame is its own level. | |
| """ | |
| chunks = [audio[s:e] for s, e in bounds] | |
| if len(chunks) < 2: | |
| return chunks | |
| frame = int(0.1 * sample_rate) | |
| floor = np.percentile(_frame_rms(audio, frame), 95) * 10 ** (QUIET_CHUNK_DB / 20) | |
| return [ | |
| chunk if len(chunk) >= frame and _frame_rms(chunk, frame).max() >= floor else chunk[:0] | |
| for chunk in chunks | |
| ] | |
| # Below this RMS (-100 dBFS) a chunk is digital silence: exact zeros, as in | |
| # edited or remixed recordings. Given one, the model answers with a memorized | |
| # training sentence ("The film was directed by the same director who directed | |
| # 'The Man with the Moustache'") -- ten such chunks cost 2.3 WER on one AMI | |
| # meeting -- so it is skipped. Quiet real speech sits near -60 dBFS. | |
| SILENCE_RMS = 1e-5 | |
| def is_silent(audio: npt.NDArray[np.float32]) -> bool: | |
| """True for digital silence (or an empty array): nothing for the model to hear.""" | |
| return audio.size == 0 or float(np.sqrt(np.mean(np.square(audio)))) < SILENCE_RMS | |
| class ASRProcessor(ProcessorMixin): | |
| """Processor for Whisper-based ASR models.""" | |
| attributes: ClassVar[list[str]] = ["feature_extractor", "tokenizer"] | |
| feature_extractor: SequenceFeatureExtractor | |
| tokenizer: PreTrainedTokenizerBase | |
| feature_extractor_class = "AutoFeatureExtractor" | |
| tokenizer_class = "AutoTokenizer" | |
| # Fallback only. The real value comes from `ASRConfig.audio_token`, which | |
| # resolves to the decoder's native placeholder where it has one (Gemma 4's | |
| # pretrained "<|audio|>") and to "<audio>" otherwise. Hardcoding the | |
| # fallback here fails silently on a native-token decoder: "<audio>" was | |
| # never added to that vocab, so it tokenizes into ordinary subwords and | |
| # the prompt ends up with zero scatter positions for N audio embeddings. | |
| AUDIO_TOKEN = "<audio>" | |
| TRANSCRIBE_PROMPT = DEFAULT_TRANSCRIBE_PROMPT | |
| def __init__( | |
| self, | |
| feature_extractor: SequenceFeatureExtractor, | |
| tokenizer: PreTrainedTokenizerBase, | |
| projector: "MLPAudioProjector | None" = None, | |
| encoder_conv_layers: list[ConvLayerSpec] | None = None, | |
| audio_token: str | None = None, | |
| lead_in_seconds: float = 0.0, | |
| ): | |
| """Initialize the ASR processor. | |
| Args: | |
| feature_extractor: Audio feature extractor (WhisperFeatureExtractor) | |
| tokenizer: Text tokenizer for the language model | |
| projector: Audio projector module (for computing output lengths) | |
| encoder_conv_layers: Conv layer specs [(pad, kernel, stride), ...] | |
| audio_token: Placeholder token scattered with audio embeddings. | |
| Must match `ASRConfig.audio_token` / `ASRModel.audio_token`; | |
| defaults to AUDIO_TOKEN. | |
| """ | |
| self.feature_extractor = feature_extractor | |
| self.tokenizer = tokenizer | |
| self.audio_token = audio_token or self.AUDIO_TOKEN | |
| self.audio_token_id = tokenizer.convert_tokens_to_ids(self.audio_token) | |
| self.projector = projector | |
| self.encoder_conv_layers = encoder_conv_layers or DEFAULT_ENCODER_CONV_LAYERS | |
| self.lead_in_seconds = float(lead_in_seconds) | |
| def _render_prompt(self, num_audio_tokens: int, text: str | None) -> torch.Tensor: | |
| """Tokenize one chat prompt carrying exactly `num_audio_tokens` placeholders.""" | |
| return render_audio_prompt( | |
| self.tokenizer, self.audio_token, num_audio_tokens, self.TRANSCRIBE_PROMPT, text | |
| ) | |
| def _stack_prompt_rows(self, rows: list[torch.Tensor]) -> tuple[torch.Tensor, torch.Tensor]: | |
| """Stack per-sample prompt rows into a batch (see `left_pad_prompt_rows`).""" | |
| return left_pad_prompt_rows(rows, self.tokenizer) | |
| def __call__(self, *args: Any, **kwargs: Any) -> BatchFeature: | |
| """Process audio and text inputs for inference; see `_process` for the arguments. | |
| `ProcessorMixin.__call__` takes `(images, text, videos, audio, ...)`; this | |
| processor takes audio first, so the arguments are forwarded unchanged to | |
| `_process`, which carries the real signature. | |
| """ | |
| return BatchFeature(data=self._process(*args, **kwargs)) | |
| def _process( | |
| self, | |
| audio: AudioInput | None = None, | |
| text: str | None = None, | |
| return_tensors: str = "pt", | |
| **kwargs: Any, | |
| ) -> dict[str, torch.Tensor]: | |
| """Process audio and text inputs for inference. | |
| Args: | |
| audio: Raw audio waveform(s). A batch gets one prompt per sample. | |
| text: Target transcription (optional, for training - but use DataCollator instead) | |
| return_tensors: Return format ("pt" for PyTorch) | |
| Returns: | |
| Dict with input_features, input_ids, attention_mask | |
| """ | |
| result: dict[str, torch.Tensor] = {} | |
| token_counts = [0] | |
| # Process audio | |
| if audio is not None: | |
| sr = getattr(self.feature_extractor, "sampling_rate", 16000) | |
| padded_audio = prepend_lead_in(audio, sr, self.lead_in_seconds) | |
| extract = cast("AudioFeatureExtractor", self.feature_extractor) | |
| audio_inputs = extract( | |
| padded_audio, | |
| sampling_rate=sr, | |
| return_attention_mask=True, | |
| return_tensors=return_tensors, | |
| **kwargs, | |
| ) | |
| result["input_features"] = audio_inputs["input_features"] | |
| result["audio_attention_mask"] = audio_inputs["attention_mask"] | |
| if self.projector is None: | |
| msg = ( | |
| "ASRProcessor needs a projector to size the audio prompt. Build it " | |
| "with ASRModel.get_processor() instead of constructing it directly." | |
| ) | |
| raise ValueError(msg) | |
| # One count per sample, from that sample's own mel length. Sizing a | |
| # single shared prompt from the batch max -- which this used to do -- | |
| # returns batch-1 `input_ids` against batch-B `input_features`, and | |
| # gives every shorter row more `<audio>` placeholders than the | |
| # projector produced for it. `masked_scatter` then mis-scatters | |
| # silently. This is the same failure `_prepare_audio_inputs` | |
| # documents as fixed on the model side, and it only shows up on a | |
| # ragged batch, so batch-1 eval never sees it. | |
| mel_lengths = audio_inputs["attention_mask"].sum(dim=-1).reshape(-1).long() | |
| encoder_lengths = compute_encoder_output_length(mel_lengths, self.encoder_conv_layers) | |
| token_counts = self.projector.get_output_length(encoder_lengths).tolist() | |
| rows = [self._render_prompt(n, text) for n in token_counts] | |
| input_ids, attention_mask = self._stack_prompt_rows(rows) | |
| result["input_ids"] = input_ids | |
| result["attention_mask"] = attention_mask | |
| return result | |
| ASRProcessor.register_for_auto_class() | |
| transformers.AutoProcessor.register(ASRConfig, ASRProcessor) | |