fix(stt): prefer cached local whisper models

This commit is contained in:
fangliquan
2026-09-15 02:13:30 +08:00
committed by Teknium
parent bae9f8ab85
commit ebb6dc6e70
5 changed files with 84 additions and 7 deletions

View File

@@ -351,6 +351,44 @@ class TestTranscribeLocalCommand:
# _transcribe_local — additional tests
# ============================================================================
class TestLocalModelLoading:
def test_cached_model_load_never_uses_online_resolution(self):
cached_model = object()
with patch("faster_whisper.WhisperModel", return_value=cached_model) as model_cls:
from tools.transcription_local import _create_whisper_model
assert _create_whisper_model("base", device="cpu", compute_type="int8") is cached_model
model_cls.assert_called_once_with(
"base", local_files_only=True, device="cpu", compute_type="int8"
)
@pytest.mark.parametrize("download_error", [None, RuntimeError("ConnectTimeout")])
def test_cache_miss_falls_back_with_actionable_download_failure(self, download_error):
from huggingface_hub.errors import LocalEntryNotFoundError
from tools.transcription_local import _create_whisper_model
downloaded_model = object()
online_result = download_error or downloaded_model
side_effect = [LocalEntryNotFoundError("not cached"), online_result]
with patch("faster_whisper.WhisperModel", side_effect=side_effect) as model_cls:
if download_error:
with pytest.raises(RuntimeError) as exc_info:
_create_whisper_model("base", device="auto", compute_type="auto")
assert "HF_ENDPOINT" in str(exc_info.value)
assert "HF_HUB_DISABLE_XET=1" in str(exc_info.value)
else:
assert _create_whisper_model(
"base", device="auto", compute_type="auto"
) is downloaded_model
assert model_cls.call_args_list == [
call("base", local_files_only=True, device="auto", compute_type="auto"),
call("base", local_files_only=False, device="auto", compute_type="auto"),
]
@pytest.mark.skipif(
not __import__("importlib").util.find_spec("faster_whisper"),
reason="faster_whisper not installed",
@@ -416,7 +454,9 @@ class TestTranscribeLocalExtended:
result = _transcribe_local(str(audio), "base")
assert result["success"] is True
mock_whisper_cls.assert_called_once_with("base", device="cpu", compute_type="float32")
mock_whisper_cls.assert_called_once_with(
"base", local_files_only=True, device="cpu", compute_type="float32"
)
def test_cuda_out_of_memory_does_not_trigger_cpu_fallback(self, tmp_path):

View File

@@ -120,6 +120,27 @@ def _get_idle_unload_seconds(local_cfg: Dict[str, Any]) -> int:
return max(_config_number(local_cfg, "unload_after_idle_seconds", 0, int), 0)
def _create_whisper_model(model_name: str, *, device: str, compute_type: str):
"""Use a cached model without contacting the Hub, downloading only on a cache miss."""
from faster_whisper import WhisperModel
from huggingface_hub.errors import LocalEntryNotFoundError
kwargs = {"device": device, "compute_type": compute_type}
try:
return WhisperModel(model_name, local_files_only=True, **kwargs)
except LocalEntryNotFoundError:
logger.info("faster-whisper model '%s' is not cached; downloading it from the Hugging Face Hub", model_name)
try:
return WhisperModel(model_name, local_files_only=False, **kwargs)
except Exception as exc:
raise RuntimeError(
f"Unable to download faster-whisper model '{model_name}': {exc}. "
"If huggingface.co is unreachable, set HF_ENDPOINT to an accessible mirror; "
"when using a mirror with hf-xet installed, also set HF_HUB_DISABLE_XET=1."
) from exc
def _load_local_whisper_model(model_name: str, device: str = "auto", compute_type: str = "auto"):
"""Load faster-whisper with graceful CUDA → CPU fallback. ``device="auto"`` picks CUDA
whenever the ctranslate2 wheel ships CUDA libs, even on hosts without the NVIDIA runtime (WSL2,
@@ -134,19 +155,18 @@ def _load_local_whisper_model(model_name: str, device: str = "auto", compute_typ
# Importing ctranslate2 can itself abort on Apple Silicon/Rosetta when
# multiple Intel OpenMP runtimes are loaded — set before the import.
os.environ.setdefault("KMP_DUPLICATE_LIB_OK", "TRUE")
from faster_whisper import WhisperModel
if force_cpu:
logger.info("Apple Silicon/Rosetta detected — loading faster-whisper on CPU "
"(int8) to avoid native device autodetection crashes")
return WhisperModel(model_name, device="cpu", compute_type="int8")
return _create_whisper_model(model_name, device="cpu", compute_type="int8")
try:
return WhisperModel(model_name, device=device, compute_type=compute_type)
return _create_whisper_model(model_name, device=device, compute_type=compute_type)
except Exception as exc:
if not _looks_like_cuda_lib_error(exc):
raise
logger.warning("faster-whisper CUDA load failed (%s) — falling back to CPU (int8). "
"Install the NVIDIA CUDA runtime (libcublas/libcudnn) to use GPU.", exc)
return WhisperModel(model_name, device="cpu", compute_type="int8")
return _create_whisper_model(model_name, device="cpu", compute_type="int8")
# Silence-hallucination hardening for local faster-whisper (whisper decodes junk like

View File

@@ -330,8 +330,7 @@ def _get_or_load_local_model(model_name: str, local_cfg: Dict[str, Any]):
def _replace_cached_model_on_cpu(model_name: str):
"""Load *model_name* on CPU/int8 and make it the cached singleton."""
global _local_model, _local_model_name
from faster_whisper import WhisperModel
model = WhisperModel(model_name, device="cpu", compute_type="int8")
model = _load_local_whisper_model(model_name, device="cpu", compute_type="int8")
with _local_model_lock:
_local_model, _local_model_name = model, model_name
return model

View File

@@ -502,6 +502,15 @@ stt:
| `medium` | ~1.5 GB | Slower | Great |
| `large-v3` | ~3 GB | Slowest | Best |
The first use downloads the selected model from `huggingface.co`; later loads prefer the local cache and do not require an online revision check. On networks where the Hub is unavailable, export an accessible mirror in the shell or service that starts Hermes:
```bash
HF_ENDPOINT=https://your-hugging-face-mirror.example
HF_HUB_DISABLE_XET=1
```
`HF_HUB_DISABLE_XET=1` keeps downloads on the mirror's regular HTTP path instead of contacting Xet CAS hosts that do not honor `HF_ENDPOINT`.
**Groq API** — Requires `GROQ_API_KEY`. Good cloud fallback when you want a free hosted STT option. Set `stt.groq.language` (or the global `HERMES_LOCAL_STT_LANGUAGE` env var) to skip Whisper's auto-detect and reduce latency on known-language audio.
**OpenAI API** — Accepts `VOICE_TOOLS_OPENAI_KEY` first and falls back to `OPENAI_API_KEY`. Supports `whisper-1`, `gpt-4o-mini-transcribe`, `gpt-4o-transcribe`, and `gpt-transcribe`.

View File

@@ -107,6 +107,15 @@ ELEVENLABS_API_KEY=*** # ElevenLabs — premium quality
If `faster-whisper` is installed, voice mode works with **zero API keys** for STT. The model (~150 MB for `base`) downloads automatically on first use.
:::
The first download normally comes from `huggingface.co`. If that host is unavailable on your network, export an accessible mirror in the shell or service that starts Hermes:
```bash
HF_ENDPOINT=https://your-hugging-face-mirror.example
HF_HUB_DISABLE_XET=1
```
Disabling Xet avoids authentication failures from Xet's separate CAS hosts when a mirror is in use. After the model is cached, Hermes loads that snapshot without an online revision check.
---
## CLI Voice Mode