From a683ef95d2cfebd266f9738d70bcccf157e91e2b Mon Sep 17 00:00:00 2001 From: kshitij Date: Mon, 3 Aug 2026 16:14:30 +0530 Subject: [PATCH] feat(stt): pre-upload silence trim for cloud providers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Local faster-whisper gets Silero VAD (bf8004e3a) so silence never reaches the model. Cloud providers got no such protection: the raw file uploads untouched, so every second of silence in a voice note is paid for twice — upload time and per-audio-minute billing — and cloud Whisper hallucinates junk tokens on silent stretches exactly like local Whisper did before the VAD hardening. A 13s voice note with two long pauses is billed as 13s of audio to transcribe ~6s of speech. Close the gap client-side: before uploading to a built-in cloud provider (groq/openai/mistral/xai/elevenlabs/deepinfra), collapse long pauses with ffmpeg's silenceremove filter, keeping stt.cloud_trim_keep_ms (default 300) of every pause so word boundaries and natural pacing survive. Uses ffmpeg, already a dependency of this exact path via _transcode_audio_for_stt — no new dependency. The trim is strictly best-effort — ALL of these upload the original untouched, transcription never fails because of the trim: - stt.cloud_trim_silence: false - ffmpeg/ffprobe missing, trim failure, or timeout - trimmed result ~empty (mostly-silence clip: the provider, not a client-side dB heuristic, decides whether it contains speech) - trim saves <10% (re-encoding for nothing) Command-type and plugin providers are deliberately NOT trimmed: they may wrap local CLIs that want the original bytes or run their own VAD. E2E (real ffmpeg + faster-whisper): 13.2s voice note with 7s pause -> 6.2s upload (-53%); transcript of trimmed audio matches the original on both utterances. Dense-speech and all-silence WAVs correctly fall back to the original. 22 unit+E2E tests; STT/voice suite failures identical to upstream/main baseline (all pre-existing). --- cli-config.yaml.example | 7 + hermes_cli/config_defaults.py | 8 + tests/tools/test_stt_cloud_trim.py | 294 +++++++++++++++++++++++ tools/transcription_tools.py | 179 ++++++++++++++ website/docs/user-guide/configuration.md | 5 + 5 files changed, 493 insertions(+) create mode 100644 tests/tools/test_stt_cloud_trim.py diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 09b33164f8..3489749a39 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -1193,6 +1193,13 @@ platform_toolsets: stt: enabled: true # provider: "local" # auto-detected if omitted + # --- Cloud pre-upload silence trim (groq/openai/mistral/xai/elevenlabs/deepinfra) --- + # Local whisper gets Silero VAD; cloud endpoints otherwise receive raw audio — + # silence inflates upload time, per-audio-minute billing, and hallucination risk. + # Collapses pauses with ffmpeg client-side; on any failure the original uploads untouched. + # cloud_trim_silence: true # set false to always upload the original audio + # cloud_trim_threshold_db: -40 # audio quieter than this counts as silence + # cloud_trim_keep_ms: 300 # how much of each pause survives (keeps natural pacing) local: model: "base" # tiny | base | small | medium | large-v3 | turbo # language: "" # auto-detect; set to "en", "es", "fr", etc. to force diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 01fe5bfe0f..13554ff68e 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1519,6 +1519,14 @@ DEFAULT_CONFIG = { # "STT transcribed the wrong language". Set to "" to restore # auto-detect, or to your language code ("es", "zh", "uk", ...). "language": "en", + # Pre-upload silence trim for cloud providers (groq/openai/mistral/ + # xai/elevenlabs/deepinfra). Local whisper gets Silero VAD; cloud + # endpoints otherwise receive raw audio — silence inflates upload + # time, per-audio-minute billing, and hallucination risk. Collapses + # pauses with ffmpeg client-side; any failure uploads the original. + "cloud_trim_silence": True, + "cloud_trim_threshold_db": -40, # audio quieter than this counts as silence + "cloud_trim_keep_ms": 300, # how much of each pause survives (keeps natural pacing) "local": { "model": "base", # tiny, base, small, medium, large-v3 "language": "", # auto-detect by default; set to "en", "es", "fr", etc. to force diff --git a/tests/tools/test_stt_cloud_trim.py b/tests/tools/test_stt_cloud_trim.py new file mode 100644 index 0000000000..fd21f53221 --- /dev/null +++ b/tests/tools/test_stt_cloud_trim.py @@ -0,0 +1,294 @@ +"""Tests for the cloud STT pre-upload silence trim. + +Local faster-whisper gets Silero VAD (``build_local_transcribe_kwargs``); +cloud providers upload the raw file. ``_trim_silence_for_cloud_stt`` +closes that gap: it collapses long pauses with ffmpeg before upload so +silence isn't uploaded, billed per audio-minute, or hallucinated on. + +Contract under test: + +1. Trim runs only for built-in CLOUD providers — never local/local_command, + never command-type or plugin providers. +2. Best-effort semantics: disabled config, missing ffmpeg/ffprobe, trim + failure, mostly-silence result, or <10% saving all mean "upload the + original untouched" (return None) — the transcription NEVER fails + because of the trim. +3. The dispatcher passes the trimmed file to the provider and cleans up + the temp dir afterwards. +4. E2E (real ffmpeg): a WAV with long silent stretches gets measurably + shorter; a fully-silent WAV falls back to the original. +""" + +import shutil +import struct +import sys +import types +import wave +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + +if "faster_whisper" not in sys.modules: + faster_whisper_stub = types.ModuleType("faster_whisper") + faster_whisper_stub.WhisperModel = MagicMock(name="WhisperModel") + from importlib.machinery import ModuleSpec + faster_whisper_stub.__spec__ = ModuleSpec("faster_whisper", loader=None) + sys.modules["faster_whisper"] = faster_whisper_stub + +from tools.transcription_tools import ( + CLOUD_STT_PROVIDERS, + BUILTIN_STT_PROVIDERS, + _cloud_trim_settings, + _CLOUD_TRIM_KEEP_MS_DEFAULT, + _CLOUD_TRIM_THRESHOLD_DB_DEFAULT, + _trim_silence_for_cloud_stt, +) + +_HAS_FFMPEG = bool(shutil.which("ffmpeg")) and bool(shutil.which("ffprobe")) + + +# ============================================================================ +# Helpers +# ============================================================================ + + +def _write_wav(path: Path, segments) -> str: + """Write a 16 kHz mono WAV from (kind, seconds) segments. + + kind is "tone" (audible square-ish wave) or "silence". + """ + rate = 16000 + frames = bytearray() + for kind, seconds in segments: + n = int(rate * seconds) + if kind == "tone": + # 400 Hz square wave at strong amplitude — unambiguous speech-band energy. + samples = [12000 if (i // 20) % 2 == 0 else -12000 for i in range(n)] + else: + samples = [0] * n + frames.extend(struct.pack(f"<{n}h", *samples)) + with wave.open(str(path), "wb") as wf: + wf.setnchannels(1) + wf.setsampwidth(2) + wf.setframerate(rate) + wf.writeframes(bytes(frames)) + return str(path) + + +# ============================================================================ +# Provider gating +# ============================================================================ + + +class TestProviderGating: + def test_cloud_set_excludes_local_providers(self): + assert "local" not in CLOUD_STT_PROVIDERS + assert "local_command" not in CLOUD_STT_PROVIDERS + + def test_cloud_set_covers_every_remote_builtin(self): + # Invariant: every built-in that is not local-ish uploads audio and + # must get the trim. New built-ins are cloud unless proven otherwise. + assert CLOUD_STT_PROVIDERS == BUILTIN_STT_PROVIDERS - {"local", "local_command"} + + def test_local_provider_never_trims(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + with patch("tools.transcription_tools._load_stt_config", + return_value={"provider": "local", "enabled": True}), \ + patch("tools.transcription_tools._trim_silence_for_cloud_stt") as trim, \ + patch("tools.transcription_tools._transcribe_local", + return_value={"success": True, "transcript": "ok"}): + from tools.transcription_tools import _transcribe_prepared_audio + result = _transcribe_prepared_audio(wav) + assert result["success"] is True + trim.assert_not_called() + + def test_cloud_provider_trims_and_forwards_trimmed_path(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + trimmed_dir = tmp_path / "trim-work" + trimmed_dir.mkdir() + trimmed = _write_wav(trimmed_dir / "a-trimmed.wav", [("tone", 1)]) + seen = {} + + def fake_groq(file_path, model_name): + seen["path"] = file_path + return {"success": True, "transcript": "hi", "provider": "groq"} + + with patch("tools.transcription_tools._load_stt_config", + return_value={"provider": "groq", "enabled": True}), \ + patch("tools.transcription_tools._get_provider", return_value="groq"), \ + patch("tools.transcription_tools._trim_silence_for_cloud_stt", + return_value=trimmed), \ + patch("tools.transcription_tools._transcribe_groq", side_effect=fake_groq): + from tools.transcription_tools import _transcribe_prepared_audio + result = _transcribe_prepared_audio(wav) + + assert result["success"] is True + assert seen["path"] == trimmed + # Dispatcher owns the cleanup of the trim temp dir. + assert not trimmed_dir.exists() + + def test_trim_returning_none_uploads_original(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + seen = {} + + def fake_groq(file_path, model_name): + seen["path"] = file_path + return {"success": True, "transcript": "hi", "provider": "groq"} + + with patch("tools.transcription_tools._load_stt_config", + return_value={"provider": "groq", "enabled": True}), \ + patch("tools.transcription_tools._get_provider", return_value="groq"), \ + patch("tools.transcription_tools._trim_silence_for_cloud_stt", + return_value=None), \ + patch("tools.transcription_tools._transcribe_groq", side_effect=fake_groq): + from tools.transcription_tools import _transcribe_prepared_audio + result = _transcribe_prepared_audio(wav) + + assert result["success"] is True + assert seen["path"] == wav + + def test_command_provider_never_trims(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + cfg = { + "provider": "mywhisper", + "enabled": True, + "providers": {"mywhisper": {"type": "command", "command": "true"}}, + } + with patch("tools.transcription_tools._load_stt_config", return_value=cfg), \ + patch("tools.transcription_tools._trim_silence_for_cloud_stt") as trim, \ + patch("tools.transcription_tools._transcribe_command_stt", + return_value={"success": True, "transcript": "ok"}): + from tools.transcription_tools import _transcribe_prepared_audio + _transcribe_prepared_audio(wav) + trim.assert_not_called() + + +# ============================================================================ +# Settings resolution +# ============================================================================ + + +class TestCloudTrimSettings: + def test_defaults(self): + enabled, threshold, keep = _cloud_trim_settings({}) + assert enabled is True + assert threshold == _CLOUD_TRIM_THRESHOLD_DB_DEFAULT + assert keep == _CLOUD_TRIM_KEEP_MS_DEFAULT + + def test_disable(self): + enabled, _, _ = _cloud_trim_settings({"cloud_trim_silence": False}) + assert enabled is False + + def test_none_means_default_on(self): + enabled, _, _ = _cloud_trim_settings({"cloud_trim_silence": None}) + assert enabled is True + + def test_custom_values(self): + enabled, threshold, keep = _cloud_trim_settings( + {"cloud_trim_threshold_db": -30, "cloud_trim_keep_ms": 500} + ) + assert enabled is True + assert threshold == -30 + assert keep == 500 + + def test_garbage_falls_back(self): + _, threshold, keep = _cloud_trim_settings( + {"cloud_trim_threshold_db": "loud", "cloud_trim_keep_ms": None} + ) + assert threshold == _CLOUD_TRIM_THRESHOLD_DB_DEFAULT + assert keep == _CLOUD_TRIM_KEEP_MS_DEFAULT + + def test_negative_keep_clamped(self): + _, _, keep = _cloud_trim_settings({"cloud_trim_keep_ms": -100}) + assert keep == 0 + + def test_non_dict_config(self): + enabled, threshold, keep = _cloud_trim_settings(None) + assert enabled is True + assert threshold == _CLOUD_TRIM_THRESHOLD_DB_DEFAULT + + +# ============================================================================ +# Best-effort fallbacks (all must return None, never raise) +# ============================================================================ + + +class TestTrimFallbacks: + def test_disabled_returns_none(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + assert _trim_silence_for_cloud_stt(wav, {"cloud_trim_silence": False}) is None + + def test_missing_ffmpeg_returns_none(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + with patch("tools.transcription_tools._find_ffmpeg_binary", return_value=None): + assert _trim_silence_for_cloud_stt(wav, {}) is None + + def test_missing_ffprobe_returns_none(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + with patch("tools.transcription_tools._find_ffmpeg_binary", return_value="/bin/ffmpeg"), \ + patch("tools.transcription_tools._find_ffprobe_binary", return_value=None): + assert _trim_silence_for_cloud_stt(wav, {}) is None + + def test_ffmpeg_failure_returns_none_and_cleans_up(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + import subprocess as sp + + def probe(path): + return 10.0 + + with patch("tools.transcription_tools._find_ffmpeg_binary", return_value="/bin/ffmpeg"), \ + patch("tools.transcription_tools._probe_audio_duration", side_effect=probe), \ + patch("tools.transcription_tools.subprocess.run", + side_effect=sp.CalledProcessError(1, "ffmpeg")): + assert _trim_silence_for_cloud_stt(wav, {}) is None + + def test_unprobeable_source_returns_none(self, tmp_path): + wav = _write_wav(tmp_path / "a.wav", [("tone", 1)]) + with patch("tools.transcription_tools._find_ffmpeg_binary", return_value="/bin/ffmpeg"), \ + patch("tools.transcription_tools._probe_audio_duration", return_value=None): + assert _trim_silence_for_cloud_stt(wav, {}) is None + + +# ============================================================================ +# E2E with real ffmpeg +# ============================================================================ + + +@pytest.mark.skipif(not _HAS_FFMPEG, reason="ffmpeg/ffprobe not installed") +class TestTrimE2E: + def test_long_pauses_are_collapsed(self, tmp_path): + # 2s speech + 6s silence + 2s speech + 4s trailing silence = 14s, + # ~10s of it silence. The trim must save well over 10%. + wav = _write_wav( + tmp_path / "pauses.wav", + [("tone", 2), ("silence", 6), ("tone", 2), ("silence", 4)], + ) + from tools.transcription_tools import _probe_audio_duration + trimmed = _trim_silence_for_cloud_stt(wav, {}) + assert trimmed is not None + try: + original = _probe_audio_duration(wav) + result = _probe_audio_duration(trimmed) + assert result is not None and original is not None + assert result < original * 0.6 # >40% shorter + assert result > 3.5 # both speech chunks survived + finally: + shutil.rmtree(Path(trimmed).parent, ignore_errors=True) + + def test_dense_speech_untouched(self, tmp_path): + # Continuous tone — nothing to trim, saving <10% → return None. + wav = _write_wav(tmp_path / "dense.wav", [("tone", 5)]) + assert _trim_silence_for_cloud_stt(wav, {}) is None + + def test_all_silence_falls_back_to_original(self, tmp_path): + # Pure silence collapses to ~nothing; the provider must decide + # "no speech", not a client-side dB heuristic → return None. + wav = _write_wav(tmp_path / "silence.wav", [("silence", 8)]) + assert _trim_silence_for_cloud_stt(wav, {}) is None + + def test_disabled_config_uploads_original(self, tmp_path): + wav = _write_wav( + tmp_path / "pauses.wav", [("tone", 2), ("silence", 6), ("tone", 2)] + ) + assert _trim_silence_for_cloud_stt(wav, {"cloud_trim_silence": False}) is None diff --git a/tools/transcription_tools.py b/tools/transcription_tools.py index 4c03fa8235..6e3fc97dc1 100644 --- a/tools/transcription_tools.py +++ b/tools/transcription_tools.py @@ -2346,6 +2346,160 @@ def _transcribe_deepinfra(file_path: str, model_name: str) -> Dict[str, Any]: ) +# --------------------------------------------------------------------------- +# Cloud pre-upload silence trim +# --------------------------------------------------------------------------- +# +# Local faster-whisper gets Silero VAD (build_local_transcribe_kwargs) so +# silence never reaches the model. Cloud providers get no such protection: +# the raw file is uploaded, so every second of silence is paid for twice — +# once in upload time and once in per-audio-minute billing — and cloud +# Whisper hallucinates junk tokens on silent stretches exactly like local +# Whisper did before the VAD hardening. +# +# Before uploading to a built-in cloud provider we collapse long pauses with +# ffmpeg's silenceremove filter, keeping ``stt.cloud_trim_keep_ms`` of every +# pause so word boundaries and natural pacing survive. The trim is purely +# best-effort — ANY of these falls back to uploading the original untouched: +# - ``stt.cloud_trim_silence: false`` +# - ffmpeg or ffprobe not installed +# - the trim command fails or times out +# - the trimmed result is suspiciously empty (mostly-silence clip — the +# provider, not a client-side heuristic, decides whether it has speech) +# - the trim saves less than ~10% (re-encoding for nothing) +# +# Command-type and plugin providers are deliberately NOT trimmed: they may +# wrap local CLIs that want the original bytes (and may run their own VAD). + +_CLOUD_TRIM_THRESHOLD_DB_DEFAULT = -40 # audio below this level counts as silence +_CLOUD_TRIM_KEEP_MS_DEFAULT = 300 # how much of each pause survives the trim +_CLOUD_TRIM_MIN_SAVING = 0.10 # use the trimmed file only when >=10% shorter +_CLOUD_TRIM_MIN_RESULT_SECONDS = 0.3 # all-silence guard: never upload ~empty audio + +# Built-in providers that upload audio to a remote API. +CLOUD_STT_PROVIDERS = frozenset(BUILTIN_STT_PROVIDERS - {"local", "local_command"}) + + +def _find_ffprobe_binary() -> Optional[str]: + return _find_binary("ffprobe") + + +def _probe_audio_duration(file_path: str) -> Optional[float]: + """Return the audio duration in seconds via ffprobe, or None.""" + ffprobe = _find_ffprobe_binary() + if not ffprobe: + return None + command = [ + ffprobe, "-v", "error", + "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", + file_path, + ] + try: + result = subprocess.run( + command, check=True, capture_output=True, text=True, + encoding="utf-8", errors="replace", timeout=30, + stdin=subprocess.DEVNULL, creationflags=windows_hide_flags(), + ) + return float(result.stdout.strip()) + except Exception: # noqa: BLE001 - probe is best-effort + return None + + +def _cloud_trim_settings(stt_config: Dict[str, Any]) -> tuple[bool, int, int]: + """Resolve (enabled, threshold_db, keep_ms) for the cloud silence trim.""" + cfg = stt_config if isinstance(stt_config, dict) else {} + enabled = cfg.get("cloud_trim_silence", True) + if enabled is None: + enabled = True + try: + threshold_db = int(cfg.get("cloud_trim_threshold_db", _CLOUD_TRIM_THRESHOLD_DB_DEFAULT)) + except (TypeError, ValueError): + threshold_db = _CLOUD_TRIM_THRESHOLD_DB_DEFAULT + try: + keep_ms = int(cfg.get("cloud_trim_keep_ms", _CLOUD_TRIM_KEEP_MS_DEFAULT)) + except (TypeError, ValueError): + keep_ms = _CLOUD_TRIM_KEEP_MS_DEFAULT + return bool(enabled), threshold_db, max(keep_ms, 0) + + +def _trim_silence_for_cloud_stt( + file_path: str, stt_config: Dict[str, Any] +) -> Optional[str]: + """Return a silence-trimmed copy of *file_path* for cloud upload, or None. + + ``None`` always means "upload the original file": the trim is disabled, + the tools are missing, the trim failed, the clip is mostly silence, or + trimming would not save enough to justify the re-encode. On success the + caller owns deleting the returned file's parent directory. + """ + enabled, threshold_db, keep_ms = _cloud_trim_settings(stt_config) + if not enabled: + return None + ffmpeg = _find_ffmpeg_binary() + if not ffmpeg: + logger.debug("Cloud STT silence trim skipped: ffmpeg not found") + return None + original_duration = _probe_audio_duration(file_path) + if not original_duration or original_duration <= 0: + logger.debug("Cloud STT silence trim skipped: could not probe %s", file_path) + return None + + keep_seconds = keep_ms / 1000.0 + # start_periods=1 strips leading silence; stop_periods=-1 collapses every + # interior/trailing silence, keeping ``keep_seconds`` of each pause. + filter_expr = ( + f"silenceremove=" + f"start_periods=1:start_threshold={threshold_db}dB:start_silence={keep_seconds}:" + f"stop_periods=-1:stop_threshold={threshold_db}dB:stop_silence={keep_seconds}" + ) + work_dir = tempfile.mkdtemp(prefix="hermes-stt-trim-") + trimmed_path = os.path.join(work_dir, f"{Path(file_path).stem or 'audio'}-trimmed.m4a") + command = [ + ffmpeg, "-y", "-i", file_path, + "-vn", "-af", filter_expr, + "-ac", "1", "-ar", "16000", + "-c:a", "aac", "-b:a", "32k", "-movflags", "+faststart", + trimmed_path, + ] + keep_result = False + try: + subprocess.run( + command, check=True, capture_output=True, text=True, + encoding="utf-8", errors="replace", timeout=120, + stdin=subprocess.DEVNULL, creationflags=windows_hide_flags(), + ) + trimmed_duration = _probe_audio_duration(trimmed_path) + if not trimmed_duration or trimmed_duration < _CLOUD_TRIM_MIN_RESULT_SECONDS: + # Mostly/all silence. Deciding "no speech" belongs to the + # provider, not a client-side dB heuristic — upload the original. + logger.debug( + "Cloud STT silence trim discarded for %s: trimmed result ~empty (%.2fs)", + Path(file_path).name, trimmed_duration or 0.0, + ) + return None + if trimmed_duration > original_duration * (1 - _CLOUD_TRIM_MIN_SAVING): + logger.debug( + "Cloud STT silence trim discarded for %s: saves <%.0f%% (%.1fs -> %.1fs)", + Path(file_path).name, _CLOUD_TRIM_MIN_SAVING * 100, + original_duration, trimmed_duration, + ) + return None + logger.info( + "Trimmed silence from %s before cloud STT upload (%.1fs -> %.1fs, -%d%%)", + Path(file_path).name, original_duration, trimmed_duration, + round((1 - trimmed_duration / original_duration) * 100), + ) + keep_result = True + return trimmed_path + except Exception as exc: # noqa: BLE001 - trim is best-effort + logger.debug("Cloud STT silence trim failed for %s: %s", file_path, exc) + return None + finally: + if not keep_result: + shutil.rmtree(work_dir, ignore_errors=True) + + # --------------------------------------------------------------------------- # Public API # --------------------------------------------------------------------------- @@ -2410,6 +2564,31 @@ def _transcribe_prepared_audio(file_path: str, model: Optional[str] = None) -> D return {"success": False, "transcript": "", "error": "CAF audio could not be converted to WAV."} + # Pre-upload silence trim for built-in cloud providers: local whisper gets + # Silero VAD, cloud endpoints get the raw file — collapse long pauses + # client-side so silence isn't uploaded, billed, or hallucinated on. + # Best-effort: any failure uploads the original untouched. + trim_cleanup_dir: Optional[str] = None + if provider in CLOUD_STT_PROVIDERS: + trimmed = _trim_silence_for_cloud_stt(file_path, stt_config) + if trimmed: + file_path = trimmed + trim_cleanup_dir = os.path.dirname(trimmed) + + try: + return _dispatch_stt_provider(file_path, provider, stt_config, model) + finally: + if trim_cleanup_dir: + shutil.rmtree(trim_cleanup_dir, ignore_errors=True) + + +def _dispatch_stt_provider( + file_path: str, + provider: str, + stt_config: Dict[str, Any], + model: Optional[str] = None, +) -> Dict[str, Any]: + """Route *file_path* to the handler for *provider* (built-in > command > plugin).""" if provider == "local": local_cfg = stt_config.get("local") or {} model_name = _normalize_local_model( diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 9db34761af..698f954bf8 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1861,6 +1861,9 @@ stt: echo_transcripts: true # Post raw transcripts back to the chat as 🎙️ "..." (default: true) provider: "local" # "local" | "groq" | "openai" | "mistral" | "xai" | "elevenlabs" | "deepinfra" | ... language: "en" # GLOBAL language hint for every provider (per-provider language wins); set "" for auto-detect + cloud_trim_silence: true # trim long pauses with ffmpeg before uploading to a cloud provider (default: true) + cloud_trim_threshold_db: -40 # audio quieter than this counts as silence + cloud_trim_keep_ms: 300 # how much of each pause survives the trim (keeps natural pacing) local: model: "base" # tiny, base, small, medium, large-v3 language: "" # per-provider override of stt.language @@ -1887,6 +1890,8 @@ Provider behavior: - `groq` uses Groq's Whisper-compatible endpoint and reads `GROQ_API_KEY`. Pass `stt.groq.language` (or the global `HERMES_LOCAL_STT_LANGUAGE` env var) to skip auto-detection and reduce latency. - `openai` uses the OpenAI speech API and reads `VOICE_TOOLS_OPENAI_KEY`. +Cloud providers (groq, openai, mistral, xai, elevenlabs, deepinfra) get a **pre-upload silence trim** by default when `ffmpeg` is installed: long pauses in a voice note are collapsed client-side before the file uploads, keeping `cloud_trim_keep_ms` of each pause so natural pacing survives. Shorter audio means faster uploads, lower per-audio-minute billing, and fewer silence hallucinations from the remote model. The trim is best-effort — if ffmpeg is missing, the trim fails, the clip is mostly silence, or trimming would save less than ~10%, the original file is uploaded untouched. Set `stt.cloud_trim_silence: false` to always upload the original (e.g. when transcribing music or ambient audio through a cloud provider). Command-type and plugin providers never get trimmed audio. + If the requested provider is unavailable, Hermes falls back automatically in this order: `local` → `groq` → `openai`. Groq and OpenAI model overrides are environment-driven: