From b35b5d8a17c0d59e80a8b3627b679c2c1003d04f Mon Sep 17 00:00:00 2001 From: drbaph <84208527+Saganaki22@users.noreply.github.com> Date: Sat, 4 Jul 2026 20:40:35 +0100 Subject: [PATCH] fix: whisper transcription compatibility with newer transformers (#274) - Use getattr for max_length to handle removed WhisperConfig attribute - Cast input_features to model dtype to fix float16 mismatch --- nodes/audio.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/nodes/audio.py b/nodes/audio.py index cb5c637..6fb7bf5 100644 --- a/nodes/audio.py +++ b/nodes/audio.py @@ -277,14 +277,14 @@ class MTB_AudioToText(MtbAudio): f"Processing chunk {chunk_offset:.1f}s - {chunk_end / sample_rate:.1f}s" ) - max_length = model.config.max_length or 448 + max_length = getattr(model.config, "max_length", None) or 448 attention_mask = torch.ones((1, max_length)) input_features = processor( chunk_waveform, sampling_rate=sample_rate, return_tensors="pt", - ).input_features.to(device) + ).input_features.to(device=device, dtype=model.dtype) with torch.no_grad(): predicted_ids = model.generate(