fix(stt): consume lazy whisper segments inside the CUDA→CPU retry guard
faster-whisper's `model.transcribe()` returns a lazy generator; ctranslate2 dlopens the CUDA runtime on the FIRST encode, which happens while the segments are iterated in `_join_confident_segments()` — outside the try/except that implements the CUDA → CPU fallback in `_transcribe_local`. On a host with an NVIDIA driver but no CUDA runtime (Windows `cublas64_12.dll`, Linux `libcublas.so.12`) the model loads fine, the error escapes the guard, and every voice note fails with "Local transcription failed: Library cublas64_12.dll is not found or cannot be loaded" until the user pins `stt.local.device: cpu`. Materialize the segments inside the guarded block (first attempt and CPU retry) so the dlopen failure reaches the existing evict-and-retry-on-CPU path. Salvaged from #103848 by @Sahilvishnaliya (earliest fix of this class). Trimmed during salvage: the `_CUDA_LIB_ERROR_MARKERS` additions (`cublas64_`, `cudnn64_`, `cudart64_`) — the Windows message already matches the existing "cannot be loaded" marker, proven by the live probe with the reporter's exact string; the 6-test file was reduced to 2 invariant tests in the existing suite. Fixes #111929 Fixes #105295 Part of #103793 (the CPU fallback now fires; GPU-wheel install is separate) Co-authored-by: atmaksri <sri.atmakur@gmail.com> Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com> Co-authored-by: isoenthusiast <287677567+isoenthusiast@users.noreply.github.com>
This commit is contained in:
@@ -498,6 +498,63 @@ class TestTranscribeLocalExtended:
|
||||
assert result["success"] is False
|
||||
assert "CUDA out of memory" in result["error"]
|
||||
|
||||
@staticmethod
|
||||
def _lazy_failure(message):
|
||||
"""faster-whisper's transcribe() returns a lazy generator: the decode — and the CUDA
|
||||
dlopen-on-first-use — only fires while segments are iterated (#103793, #105295, #111929)."""
|
||||
def segments():
|
||||
raise RuntimeError(message)
|
||||
yield # pragma: no cover
|
||||
return segments()
|
||||
|
||||
def test_iteration_time_cuda_dlopen_retries_on_cpu(self, tmp_path):
|
||||
"""A missing CUDA library raised while ITERATING segments must evict the cached model and
|
||||
retry on CPU/int8, not surface as a hard failure (Windows: cublas64_12.dll)."""
|
||||
audio = tmp_path / "test.ogg"
|
||||
audio.write_bytes(b"fake")
|
||||
info = MagicMock(language="en", duration=1.0)
|
||||
|
||||
cuda_model = MagicMock()
|
||||
cuda_model.transcribe.return_value = (
|
||||
self._lazy_failure("Library cublas64_12.dll is not found or cannot be loaded"), info)
|
||||
cpu_segment = MagicMock(text="hi", no_speech_prob=0.0, avg_logprob=0.0)
|
||||
cpu_model = MagicMock()
|
||||
cpu_model.transcribe.return_value = ([cpu_segment], info)
|
||||
mock_whisper_cls = MagicMock(side_effect=[cuda_model, cpu_model])
|
||||
|
||||
with patch("tools.transcription_tools._HAS_FASTER_WHISPER", True), \
|
||||
patch("faster_whisper.WhisperModel", mock_whisper_cls), \
|
||||
patch("tools.transcription_tools._local_model", None), \
|
||||
patch("tools.transcription_tools._local_model_name", None):
|
||||
from tools.transcription_tools import _transcribe_local
|
||||
result = _transcribe_local(str(audio), "base")
|
||||
|
||||
assert result["success"] is True, result.get("error")
|
||||
assert result["transcript"] == "hi"
|
||||
assert mock_whisper_cls.call_count == 2
|
||||
retry_kwargs = mock_whisper_cls.call_args_list[1].kwargs
|
||||
assert (retry_kwargs["device"], retry_kwargs["compute_type"]) == ("cpu", "int8")
|
||||
|
||||
def test_iteration_time_non_lib_error_surfaces_without_cpu_retry(self, tmp_path):
|
||||
"""A real runtime failure during iteration must NOT trigger the CPU retry."""
|
||||
audio = tmp_path / "test.ogg"
|
||||
audio.write_bytes(b"fake")
|
||||
|
||||
cuda_model = MagicMock()
|
||||
cuda_model.transcribe.return_value = (self._lazy_failure("CUDA out of memory"), MagicMock())
|
||||
mock_whisper_cls = MagicMock(return_value=cuda_model)
|
||||
|
||||
with patch("tools.transcription_tools._HAS_FASTER_WHISPER", True), \
|
||||
patch("faster_whisper.WhisperModel", mock_whisper_cls), \
|
||||
patch("tools.transcription_tools._local_model", None), \
|
||||
patch("tools.transcription_tools._local_model_name", None):
|
||||
from tools.transcription_tools import _transcribe_local
|
||||
result = _transcribe_local(str(audio), "base")
|
||||
|
||||
assert result["success"] is False
|
||||
assert "CUDA out of memory" in result["error"]
|
||||
assert mock_whisper_cls.call_count == 1
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# Model auto-correction
|
||||
|
||||
@@ -356,6 +356,12 @@ def _transcribe_local(
|
||||
if v})
|
||||
try:
|
||||
segments, info = model.transcribe(file_path, **transcribe_kwargs)
|
||||
# faster-whisper's transcribe() is lazy: the decode (and with it the
|
||||
# dlopen-on-first-use of the CUDA runtime on Windows) happens while
|
||||
# ITERATING segments, after this call has already returned (#103793).
|
||||
# Consume inside the guard so a first-use cuBLAS/cuDNN load failure
|
||||
# retries on CPU exactly like a load-time failure does.
|
||||
segments = list(segments)
|
||||
except Exception as exc:
|
||||
# CUDA libs can fail at dlopen-on-first-use, AFTER loading: evict the poisoned
|
||||
# cached model, reload on CPU and retry once, else every later message fails.
|
||||
@@ -365,6 +371,7 @@ def _transcribe_local(
|
||||
"evicting cached model and retrying on CPU (int8).", exc)
|
||||
model = _replace_cached_model_on_cpu(model_name)
|
||||
segments, info = model.transcribe(file_path, **transcribe_kwargs)
|
||||
segments = list(segments)
|
||||
transcript = _join_confident_segments(segments, local_cfg)
|
||||
logger.info("Transcribed %s via local whisper (%s, lang=%s, %.1fs audio)",
|
||||
Path(file_path).name, model_name, info.language, info.duration)
|
||||
|
||||
Reference in New Issue
Block a user