From afe9e25c57904fc83e31f998c1ac9ea54d1a87f6 Mon Sep 17 00:00:00 2001 From: Sahil Vishnalya <222165401+Sahilvishnaliya@users.noreply.github.com> Date: Tue, 15 Sep 2026 11:41:24 -0700 Subject: [PATCH] =?UTF-8?q?fix(stt):=20consume=20lazy=20whisper=20segments?= =?UTF-8?q?=20inside=20the=20CUDA=E2=86=92CPU=20retry=20guard?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit faster-whisper's `model.transcribe()` returns a lazy generator; ctranslate2 dlopens the CUDA runtime on the FIRST encode, which happens while the segments are iterated in `_join_confident_segments()` — outside the try/except that implements the CUDA → CPU fallback in `_transcribe_local`. On a host with an NVIDIA driver but no CUDA runtime (Windows `cublas64_12.dll`, Linux `libcublas.so.12`) the model loads fine, the error escapes the guard, and every voice note fails with "Local transcription failed: Library cublas64_12.dll is not found or cannot be loaded" until the user pins `stt.local.device: cpu`. Materialize the segments inside the guarded block (first attempt and CPU retry) so the dlopen failure reaches the existing evict-and-retry-on-CPU path. Salvaged from #103848 by @Sahilvishnaliya (earliest fix of this class). Trimmed during salvage: the `_CUDA_LIB_ERROR_MARKERS` additions (`cublas64_`, `cudnn64_`, `cudart64_`) — the Windows message already matches the existing "cannot be loaded" marker, proven by the live probe with the reporter's exact string; the 6-test file was reduced to 2 invariant tests in the existing suite. Fixes #111929 Fixes #105295 Part of #103793 (the CPU fallback now fires; GPU-wheel install is separate) Co-authored-by: atmaksri Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com> Co-authored-by: isoenthusiast <287677567+isoenthusiast@users.noreply.github.com> --- tests/tools/test_transcription_tools.py | 57 +++++++++++++++++++++++++ tools/transcription_tools.py | 7 +++ 2 files changed, 64 insertions(+) diff --git a/tests/tools/test_transcription_tools.py b/tests/tools/test_transcription_tools.py index 7db1f3c730..2b181987a8 100644 --- a/tests/tools/test_transcription_tools.py +++ b/tests/tools/test_transcription_tools.py @@ -498,6 +498,63 @@ class TestTranscribeLocalExtended: assert result["success"] is False assert "CUDA out of memory" in result["error"] + @staticmethod + def _lazy_failure(message): + """faster-whisper's transcribe() returns a lazy generator: the decode — and the CUDA + dlopen-on-first-use — only fires while segments are iterated (#103793, #105295, #111929).""" + def segments(): + raise RuntimeError(message) + yield # pragma: no cover + return segments() + + def test_iteration_time_cuda_dlopen_retries_on_cpu(self, tmp_path): + """A missing CUDA library raised while ITERATING segments must evict the cached model and + retry on CPU/int8, not surface as a hard failure (Windows: cublas64_12.dll).""" + audio = tmp_path / "test.ogg" + audio.write_bytes(b"fake") + info = MagicMock(language="en", duration=1.0) + + cuda_model = MagicMock() + cuda_model.transcribe.return_value = ( + self._lazy_failure("Library cublas64_12.dll is not found or cannot be loaded"), info) + cpu_segment = MagicMock(text="hi", no_speech_prob=0.0, avg_logprob=0.0) + cpu_model = MagicMock() + cpu_model.transcribe.return_value = ([cpu_segment], info) + mock_whisper_cls = MagicMock(side_effect=[cuda_model, cpu_model]) + + with patch("tools.transcription_tools._HAS_FASTER_WHISPER", True), \ + patch("faster_whisper.WhisperModel", mock_whisper_cls), \ + patch("tools.transcription_tools._local_model", None), \ + patch("tools.transcription_tools._local_model_name", None): + from tools.transcription_tools import _transcribe_local + result = _transcribe_local(str(audio), "base") + + assert result["success"] is True, result.get("error") + assert result["transcript"] == "hi" + assert mock_whisper_cls.call_count == 2 + retry_kwargs = mock_whisper_cls.call_args_list[1].kwargs + assert (retry_kwargs["device"], retry_kwargs["compute_type"]) == ("cpu", "int8") + + def test_iteration_time_non_lib_error_surfaces_without_cpu_retry(self, tmp_path): + """A real runtime failure during iteration must NOT trigger the CPU retry.""" + audio = tmp_path / "test.ogg" + audio.write_bytes(b"fake") + + cuda_model = MagicMock() + cuda_model.transcribe.return_value = (self._lazy_failure("CUDA out of memory"), MagicMock()) + mock_whisper_cls = MagicMock(return_value=cuda_model) + + with patch("tools.transcription_tools._HAS_FASTER_WHISPER", True), \ + patch("faster_whisper.WhisperModel", mock_whisper_cls), \ + patch("tools.transcription_tools._local_model", None), \ + patch("tools.transcription_tools._local_model_name", None): + from tools.transcription_tools import _transcribe_local + result = _transcribe_local(str(audio), "base") + + assert result["success"] is False + assert "CUDA out of memory" in result["error"] + assert mock_whisper_cls.call_count == 1 + # ============================================================================ # Model auto-correction diff --git a/tools/transcription_tools.py b/tools/transcription_tools.py index 18c215fbb7..715ddbff56 100644 --- a/tools/transcription_tools.py +++ b/tools/transcription_tools.py @@ -356,6 +356,12 @@ def _transcribe_local( if v}) try: segments, info = model.transcribe(file_path, **transcribe_kwargs) + # faster-whisper's transcribe() is lazy: the decode (and with it the + # dlopen-on-first-use of the CUDA runtime on Windows) happens while + # ITERATING segments, after this call has already returned (#103793). + # Consume inside the guard so a first-use cuBLAS/cuDNN load failure + # retries on CPU exactly like a load-time failure does. + segments = list(segments) except Exception as exc: # CUDA libs can fail at dlopen-on-first-use, AFTER loading: evict the poisoned # cached model, reload on CPU and retry once, else every later message fails. @@ -365,6 +371,7 @@ def _transcribe_local( "evicting cached model and retrying on CPU (int8).", exc) model = _replace_cached_model_on_cpu(model_name) segments, info = model.transcribe(file_path, **transcribe_kwargs) + segments = list(segments) transcript = _join_confident_segments(segments, local_cfg) logger.info("Transcribed %s via local whisper (%s, lang=%s, %.1fs audio)", Path(file_path).name, model_name, info.language, info.duration)