fix(stt): consume lazy whisper segments inside the CUDA→CPU retry guard

faster-whisper's `model.transcribe()` returns a lazy generator; ctranslate2
dlopens the CUDA runtime on the FIRST encode, which happens while the segments
are iterated in `_join_confident_segments()` — outside the try/except that
implements the CUDA → CPU fallback in `_transcribe_local`. On a host with an
NVIDIA driver but no CUDA runtime (Windows `cublas64_12.dll`, Linux
`libcublas.so.12`) the model loads fine, the error escapes the guard, and every
voice note fails with "Local transcription failed: Library cublas64_12.dll is
not found or cannot be loaded" until the user pins `stt.local.device: cpu`.

Materialize the segments inside the guarded block (first attempt and CPU retry)
so the dlopen failure reaches the existing evict-and-retry-on-CPU path.

Salvaged from #103848 by @Sahilvishnaliya (earliest fix of this class).
Trimmed during salvage: the `_CUDA_LIB_ERROR_MARKERS` additions (`cublas64_`,
`cudnn64_`, `cudart64_`) — the Windows message already matches the existing
"cannot be loaded" marker, proven by the live probe with the reporter's exact
string; the 6-test file was reduced to 2 invariant tests in the existing suite.

Fixes #111929
Fixes #105295
Part of #103793 (the CPU fallback now fires; GPU-wheel install is separate)

Co-authored-by: atmaksri <sri.atmakur@gmail.com>
Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
Co-authored-by: isoenthusiast <287677567+isoenthusiast@users.noreply.github.com>
This commit is contained in:
Sahil Vishnalya
2026-09-15 11:41:24 -07:00
committed by Teknium
parent 0ff9941b76
commit afe9e25c57
2 changed files with 64 additions and 0 deletions

View File

@@ -498,6 +498,63 @@ class TestTranscribeLocalExtended:
assert result["success"] is False
assert "CUDA out of memory" in result["error"]
@staticmethod
def _lazy_failure(message):
"""faster-whisper's transcribe() returns a lazy generator: the decode — and the CUDA
dlopen-on-first-use — only fires while segments are iterated (#103793, #105295, #111929)."""
def segments():
raise RuntimeError(message)
yield # pragma: no cover
return segments()
def test_iteration_time_cuda_dlopen_retries_on_cpu(self, tmp_path):
"""A missing CUDA library raised while ITERATING segments must evict the cached model and
retry on CPU/int8, not surface as a hard failure (Windows: cublas64_12.dll)."""
audio = tmp_path / "test.ogg"
audio.write_bytes(b"fake")
info = MagicMock(language="en", duration=1.0)
cuda_model = MagicMock()
cuda_model.transcribe.return_value = (
self._lazy_failure("Library cublas64_12.dll is not found or cannot be loaded"), info)
cpu_segment = MagicMock(text="hi", no_speech_prob=0.0, avg_logprob=0.0)
cpu_model = MagicMock()
cpu_model.transcribe.return_value = ([cpu_segment], info)
mock_whisper_cls = MagicMock(side_effect=[cuda_model, cpu_model])
with patch("tools.transcription_tools._HAS_FASTER_WHISPER", True), \
patch("faster_whisper.WhisperModel", mock_whisper_cls), \
patch("tools.transcription_tools._local_model", None), \
patch("tools.transcription_tools._local_model_name", None):
from tools.transcription_tools import _transcribe_local
result = _transcribe_local(str(audio), "base")
assert result["success"] is True, result.get("error")
assert result["transcript"] == "hi"
assert mock_whisper_cls.call_count == 2
retry_kwargs = mock_whisper_cls.call_args_list[1].kwargs
assert (retry_kwargs["device"], retry_kwargs["compute_type"]) == ("cpu", "int8")
def test_iteration_time_non_lib_error_surfaces_without_cpu_retry(self, tmp_path):
"""A real runtime failure during iteration must NOT trigger the CPU retry."""
audio = tmp_path / "test.ogg"
audio.write_bytes(b"fake")
cuda_model = MagicMock()
cuda_model.transcribe.return_value = (self._lazy_failure("CUDA out of memory"), MagicMock())
mock_whisper_cls = MagicMock(return_value=cuda_model)
with patch("tools.transcription_tools._HAS_FASTER_WHISPER", True), \
patch("faster_whisper.WhisperModel", mock_whisper_cls), \
patch("tools.transcription_tools._local_model", None), \
patch("tools.transcription_tools._local_model_name", None):
from tools.transcription_tools import _transcribe_local
result = _transcribe_local(str(audio), "base")
assert result["success"] is False
assert "CUDA out of memory" in result["error"]
assert mock_whisper_cls.call_count == 1
# ============================================================================
# Model auto-correction

View File

@@ -356,6 +356,12 @@ def _transcribe_local(
if v})
try:
segments, info = model.transcribe(file_path, **transcribe_kwargs)
# faster-whisper's transcribe() is lazy: the decode (and with it the
# dlopen-on-first-use of the CUDA runtime on Windows) happens while
# ITERATING segments, after this call has already returned (#103793).
# Consume inside the guard so a first-use cuBLAS/cuDNN load failure
# retries on CPU exactly like a load-time failure does.
segments = list(segments)
except Exception as exc:
# CUDA libs can fail at dlopen-on-first-use, AFTER loading: evict the poisoned
# cached model, reload on CPU and retry once, else every later message fails.
@@ -365,6 +371,7 @@ def _transcribe_local(
"evicting cached model and retrying on CPU (int8).", exc)
model = _replace_cached_model_on_cpu(model_name)
segments, info = model.transcribe(file_path, **transcribe_kwargs)
segments = list(segments)
transcript = _join_confident_segments(segments, local_cfg)
logger.info("Transcribed %s via local whisper (%s, lang=%s, %.1fs audio)",
Path(file_path).name, model_name, info.language, info.duration)