From e61c85365bf8351a1b4f43a580e730ca01dd0a83 Mon Sep 17 00:00:00 2001 From: nox Date: Tue, 5 May 2026 20:20:30 +0200 Subject: [PATCH] fix(backend): enable long-form Whisper transcription on PyTorch path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The PyTorch Whisper transcribe() called the HF processor without truncation=False and model.generate() without return_timestamps=True. With those defaults, WhisperFeatureExtractor silently truncates inputs to 30 s (Whisper's native receptive field), so any dictation longer than ~30 s lost its tail. Setting truncation=False + padding="longest" + return_attention_mask=True on the processor, then forwarding the attention mask plus return_timestamps=True to generate(), flips HF Whisper into long-form mode: autoregressive decoding over rolling 30 s windows. Verified by round-tripping a 56.6 s Kokoro TTS sample through /transcribe — full text returned including content past the 30 s mark; previously the transcript was cut off roughly halfway through. MLX backend (mlx_backend.py) is intentionally unchanged: mlx_audio.stt's generate() already implements rolling-window long-form transcription with condition_on_previous_text in the upstream library, so it does not have the same bug. The HF-only kwargs added here would also break the MLX call signature. Co-Authored-By: Claude Opus 4.7 --- backend/backends/pytorch_backend.py | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/backend/backends/pytorch_backend.py b/backend/backends/pytorch_backend.py index faca0d6a..3e1db396 100644 --- a/backend/backends/pytorch_backend.py +++ b/backend/backends/pytorch_backend.py @@ -343,11 +343,18 @@ class PyTorchSTTBackend: # state — forcing offline here (issue #462) broke online users # whose `get_decoder_prompt_ids` / tokenizer calls issue # legitimate metadata lookups. - # Process audio + # Process audio. + # truncation=False + padding="longest" + return_attention_mask=True + # are required for long-form transcription. Without them the + # feature extractor silently truncates to 30s (Whisper's native + # window) and audio past that point is dropped. inputs = self.processor( audio, sampling_rate=16000, return_tensors="pt", + truncation=False, + padding="longest", + return_attention_mask=True, ) inputs = inputs.to(self.device) @@ -364,6 +371,8 @@ class PyTorchSTTBackend: with torch.no_grad(): predicted_ids = self.model.generate( inputs["input_features"], + attention_mask=inputs["attention_mask"], + return_timestamps=True, **generate_kwargs, )