From e7f03d40356bfdb2aedd186786155358cb1647db Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 23:48:41 -0700 Subject: [PATCH] refactor(hermes_cli/voice): split speak_text/stop_continuous/on_silence tails into phase helpers, compact rationale --- hermes_cli/voice.py | 413 ++++++++++++++++++-------------------------- 1 file changed, 172 insertions(+), 241 deletions(-) diff --git a/hermes_cli/voice.py b/hermes_cli/voice.py index 58b5606c3a..4737c911f8 100644 --- a/hermes_cli/voice.py +++ b/hermes_cli/voice.py @@ -13,16 +13,13 @@ import threading import time from typing import Any, Callable, Optional -# Modifier aliases mirrored from the TUI parser (``ui-tui/src/lib/platform.ts`` -# ``_MOD_ALIASES``) so one config value binds the same shortcut in both runtimes. -# ``super``/``win``/``windows`` are deliberately absent: prompt_toolkit has no -# super/meta modifier, so those spellings are TUI-only and normalize to the -# default (a silent fallback beats a hard startup crash; the CLI binding site -# in cli.py warns when it fires). +# Modifier aliases mirrored from the TUI parser (``ui-tui/src/lib/platform.ts`` ``_MOD_ALIASES``) +# so one config value binds the same shortcut in both runtimes. ``super``/``win``/``windows`` are +# deliberately absent: prompt_toolkit has no super/meta modifier, so those spellings are TUI-only +# and normalize to the default (a silent fallback beats a hard startup crash; cli.py warns). _VOICE_MOD_ALIASES = {"ctrl": "c-", "control": "c-", "alt": "a-", "option": "a-", "opt": "a-"} -# Named keys prompt_toolkit accepts in ``c-`` / ``a-`` form; aliases -# collapse to prompt_toolkit's canonical spelling. +# Named keys prompt_toolkit accepts as ``c-`` / ``a-``; aliases collapse to canonical. _VOICE_NAMED_KEYS = { "space": "space", "spc": "space", "enter": "enter", "return": "enter", "ret": "enter", @@ -30,26 +27,19 @@ _VOICE_NAMED_KEYS = { "backspace": "backspace", "bs": "backspace", "delete": "delete", "del": "delete", } -# ``useInputHandlers()`` intercepts ctrl+c/d/l (interrupt/quit/clear) before the -# voice check runs, so such a binding would be advertised but never fire — -# same blocklist as the TUI parser. +# ``useInputHandlers()`` intercepts ctrl+c/d/l (interrupt/quit/clear) before the voice check, so +# such a binding would be advertised but never fire (same blocklist as the TUI parser). On macOS +# the CLI's copy/exit/clear bindings also claim ``a-c``/``a-d``/``a-l`` (hermes-ink reports Alt as +# ``key.meta``), mirroring the TUI's darwin-only reservation. _VOICE_RESERVED_CTRL_CHARS = frozenset({"c", "d", "l"}) - -# On macOS the CLI's copy/exit/clear bindings also claim ``a-c``/``a-d``/``a-l`` -# and hermes-ink reports Alt as ``key.meta``; mirror the TUI's darwin-only -# reservation so ``option+c`` etc. don't bind in the CLI while the TUI falls back. _VOICE_RESERVED_ALT_CHARS_MAC = frozenset({"c", "d", "l"}) _DEFAULT_PT_KEY = "c-b" def voice_record_key_from_config(cfg: Any) -> Any: - """Shape-safe ``cfg.voice.record_key`` lookup. - - ``load_config()`` preserves scalar overrides, so a hand-edited ``voice: true`` / - ``voice: cmd+b`` leaves ``cfg["voice"]`` as a bool/str and the naive ``.get`` chain - would raise before voice could start. - """ + """Shape-safe ``cfg.voice.record_key``: a hand-edited ``voice: true`` / ``voice: cmd+b`` leaves + ``cfg["voice"]`` as a bool/str and the naive ``.get`` chain would raise before voice starts.""" voice = cfg.get("voice") if isinstance(cfg, dict) else None return voice.get("record_key") if isinstance(voice, dict) else None @@ -57,29 +47,21 @@ def voice_record_key_from_config(cfg: Any) -> Any: def normalize_voice_record_key_for_prompt_toolkit(raw: Any) -> str: """Coerce ``voice.record_key`` into prompt_toolkit's ``c-x`` / ``a-x`` format. - Mirrors the TUI parser contract (``ui-tui/src/lib/platform.ts``): non-string / empty / - typo'd / bare-char / multi-modifier / reserved ``ctrl+c|d|l`` / ``super``-family → the - documented default ``c-b``; ``ctrl+o`` → ``c-o``; named keys collapse to canonical - spelling (``ctrl+return`` → ``c-enter``). + Mirrors the TUI parser contract (``ui-tui/src/lib/platform.ts``): non-string / empty / typo'd / + bare-char / multi-modifier / reserved ``ctrl+c|d|l`` / ``super``-family → the documented default + ``c-b``; named keys collapse to canonical spelling (``ctrl+return`` → ``c-enter``). Exactly one + modifier: multi-modifier chords bind different shortcuts in prompt_toolkit (a-c-r) and + hermes-ink rejects them; a bare key is refused by the TUI parser. """ if not isinstance(raw, str): return _DEFAULT_PT_KEY - parts = [p.strip() for p in raw.strip().lower().split("+") if p.strip()] - # Exactly ``modifier+key``: multi-modifier chords bind different shortcuts in - # prompt_toolkit (a-c-r) and hermes-ink rejects them; a bare key is refused - # by the TUI parser. Both collapse to the default so the runtimes agree. if len(parts) != 2: return _DEFAULT_PT_KEY - modifier_token, key_token = parts - if modifier_token in {"super", "win", "windows"}: - return _DEFAULT_PT_KEY - normalized_mod = _VOICE_MOD_ALIASES.get(modifier_token) if not normalized_mod: return _DEFAULT_PT_KEY - if len(key_token) == 1: reserved = ( _VOICE_RESERVED_CTRL_CHARS if normalized_mod == "c-" @@ -87,9 +69,8 @@ def normalize_voice_record_key_for_prompt_toolkit(raw: Any) -> str: else frozenset() ) return _DEFAULT_PT_KEY if key_token in reserved else f"{normalized_mod}{key_token}" - - # Multi-char token must be a known named key; ``ctrl+spcae`` must not pass - # through as ``c-spcae`` (prompt_toolkit would reject it). + # Multi-char token must be a known named key; ``ctrl+spcae`` must not pass through as + # ``c-spcae`` (prompt_toolkit would reject it). named = _VOICE_NAMED_KEYS.get(key_token) return f"{normalized_mod}{named}" if named else _DEFAULT_PT_KEY @@ -102,11 +83,8 @@ def pt_key_to_sequence(pt_key: str) -> tuple[str, ...]: def format_voice_record_key_for_status(raw: Any) -> str: - """Render ``voice.record_key`` for ``/voice status`` as ``Ctrl+B`` / ``Alt+Space``. - - Mirrors the TUI's ``formatVoiceRecordKey``; malformed configs surface as the default so - status never advertises a shortcut that won't bind. - """ + """Render ``voice.record_key`` for ``/voice status`` as ``Ctrl+B`` / ``Alt+Space``; malformed + configs surface as the default so status never advertises a shortcut that won't bind.""" normalized = normalize_voice_record_key_for_prompt_toolkit(raw) prefix = "Alt+" if normalized.startswith("a-") else "Ctrl+" key = normalized[2:] @@ -125,12 +103,9 @@ logger = logging.getLogger(__name__) def _debug(msg: str) -> None: - """Debug breadcrumb when HERMES_VOICE_DEBUG=1. - - Goes to stderr so the TUI gateway surfaces it as a gateway.stderr Activity line. Broken-pipe - errors are swallowed: this fires from background threads where a dead stderr must not kill - the gateway (the stdin/stdout command pipe is what matters). - """ + """HERMES_VOICE_DEBUG=1 breadcrumb on stderr (the TUI gateway shows it as a gateway.stderr + Activity line). Broken pipes are swallowed: this fires from background threads where a dead + stderr must not kill the gateway — the stdin/stdout command pipe is what matters.""" if os.environ.get("HERMES_VOICE_DEBUG", "").strip() == "1": with contextlib.suppress(BrokenPipeError, OSError): print(f"[voice] {msg}", file=sys.stderr, flush=True) @@ -152,10 +127,8 @@ def _beeps_enabled() -> bool: def _play_beep(frequency: int, count: int = 1) -> None: - """Audible cue matching cli.py's beeps: 880 Hz once on start, 660 Hz twice on stop. - - Best-effort — a missing speaker must never break the voice loop. - """ + """Audible cue matching cli.py's beeps (880 Hz once on start, 660 Hz twice on stop). + Best-effort — a missing speaker must never break the voice loop.""" if not _beeps_enabled(): return try: @@ -167,11 +140,8 @@ def _play_beep(frequency: int, count: int = 1) -> None: def _safe_call(cb: Optional[Callable], *args: Any, warn: Optional[str] = None) -> None: - """Invoke an optional callback, swallowing its exceptions. - - ``warn`` is a ``logger.warning`` format with one ``%s`` slot for the exception; without it - failures are silently ignored (status/limit callbacks are fire-and-forget). - """ + """Invoke an optional callback, swallowing its exceptions. ``warn`` is a ``logger.warning`` + format with one ``%s`` slot; without it failures are silent (status callbacks are fire-and-forget).""" if not cb: return try: @@ -214,6 +184,7 @@ def _deactivate(on_status: Optional[Callable[[str], None]] = None) -> None: _continuous_active = False _safe_call(on_status, "idle") + # ── Push-to-talk state ─────────────────────────────────────────────── _recorder = None _recorder_lock = threading.Lock() @@ -225,38 +196,29 @@ _continuous_stopping = False _continuous_auto_restart: bool = True _continuous_recorder: Any = None -# ── TTS-vs-STT feedback guard ──────────────────────────────────────── -# TTS over the speakers lands in the live mic and gets transcribed as user input -# — an infinite loop the agent happily joins. Mirrors cli.py:_voice_tts_done: -# cleared while speak_text plays, set while silent. _continuous_on_silence waits -# on it before re-arming; speak_text cancels live capture before playback so the -# previous utterance's tail doesn't leak into the mic. +# TTS-vs-STT feedback guard: TTS over the speakers lands in the live mic and gets transcribed as +# user input — an infinite loop the agent happily joins. Mirrors cli.py:_voice_tts_done: cleared +# while speak_text plays, set while silent. The silence callback waits on it before re-arming; +# speak_text cancels live capture before playback so the previous utterance's tail doesn't leak. _tts_playing = threading.Event() _tts_playing.set() # initially "not playing" -# ── Silence-count hold (agent busy) ────────────────────────────────── -# While the agent is mid-turn (possibly minutes) or TTS plays, the user is -# CORRECTLY silent — those cycles must not count toward the no-speech limit or a -# long tool run ends the voice chat under the user. The host surface registers a -# probe reporting "agent busy"; TTS is tracked via _tts_playing. +# Silence-count hold: while the agent is mid-turn (possibly minutes) or TTS plays, the user is +# CORRECTLY silent — those cycles must not count toward the no-speech limit or a long tool run +# ends the voice chat under the user. The host surface registers a probe reporting "agent busy". _voice_busy_probe: Optional[Callable[[], bool]] = None def set_voice_busy_probe(probe: Optional[Callable[[], bool]]) -> None: """Register a callable returning True while the agent is mid-turn; ``None`` clears it. - - Must be cheap and thread-safe — it runs on the silence-callback thread. - """ + Must be cheap and thread-safe — it runs on the silence-callback thread.""" global _voice_busy_probe _voice_busy_probe = probe def _voice_activity_held() -> bool: - """True while silent cycles must NOT count toward the no-speech limit. - - Held when TTS is playing or the busy probe reports the agent mid-turn. Fail-open to "not - held" so a broken probe can never make the voice chat immortal. - """ + """True while silent cycles must NOT count toward the no-speech limit (TTS playing or agent + mid-turn). Fail-open to "not held" so a broken probe can never make the voice chat immortal.""" if not _tts_playing.is_set(): return True probe = _voice_busy_probe @@ -271,9 +233,9 @@ def _voice_activity_held() -> bool: _continuous_on_transcript: Optional[Callable[[str], None]] = None _continuous_on_status: Optional[Callable[[str], None]] = None _continuous_on_silent_limit: Optional[Callable[[], None]] = None -# Explicit user-intent stop: fired when the user SAYS a bare stop phrase. Distinct -# from on_silent_limit (a timeout) so consumers end the conversation like a manual -# stop instead of reporting "no speech detected"; unset → on_silent_limit fires. +# Explicit user-intent stop: fired when the user SAYS a bare stop phrase. Distinct from +# on_silent_limit (a timeout) so consumers end the conversation like a manual stop instead of +# reporting "no speech detected"; unset → on_silent_limit fires. _continuous_on_stop_phrase: Optional[Callable[[str], None]] = None _continuous_no_speech_count = 0 _CONTINUOUS_NO_SPEECH_LIMIT = 3 @@ -284,12 +246,15 @@ def _callbacks() -> tuple: return _continuous_on_transcript, _continuous_on_status, _continuous_on_silent_limit, _continuous_on_stop_phrase -def _detect_stop_phrase(transcript: Optional[str], where: str, tail: str) -> tuple[Optional[str], bool, str]: - """Split a transcript into (deliverable text, is_stop_phrase, stop_text). +def _turn_transcript( + wav_path: Optional[str], fail_msg: str, where: str, tail: str, debug_prefix: Optional[str] = None +) -> tuple[Optional[str], bool, str]: + """Transcribe a finished capture → (deliverable text, is_stop_phrase, stop_text). A bare stop phrase ("stop") is explicit user intent to end the voice chat: it is never sent to the agent, so the deliverable text becomes None. """ + transcript = _transcribe_wav(wav_path, fail_msg, debug_prefix) if wav_path else None if not (transcript and is_voice_stop_phrase(transcript)): return transcript, False, "" _debug(f"{where}: stop phrase {transcript!r} — {tail}") @@ -305,11 +270,9 @@ def _signal_halt(stop_phrase: bool, stop_text: str, on_stop_phrase, on_silent_li def _tally_silence(spoke: bool, held: bool, where: str) -> tuple[bool, int]: - """Update the no-speech counter (caller holds ``_continuous_lock``). - - Speech resets it; a held cycle (agent busy / TTS playing) is ignored; otherwise it bumps. - Returns (limit_hit, count_after_bump); the counter is reset when the limit is hit. - """ + """Update the no-speech counter (caller holds ``_continuous_lock``): speech resets it, a held + cycle (agent busy / TTS playing) is ignored, otherwise it bumps. Returns (limit_hit, + count_after_bump); the counter is reset when the limit is hit.""" global _continuous_no_speech_count if spoke: _continuous_no_speech_count = 0 @@ -331,7 +294,6 @@ def _tally_silence(spoke: bool, held: bool, where: str) -> tuple[bool, int]: def start_recording() -> None: """Begin capturing from the default input device (push-to-talk).""" global _recorder - with _recorder_lock: if _recorder is not None and getattr(_recorder, "is_recording", False): return @@ -343,18 +305,13 @@ def start_recording() -> None: def stop_and_transcribe() -> Optional[str]: """Stop the active push-to-talk recording, transcribe, return text.""" global _recorder - with _recorder_lock: rec = _recorder _recorder = None - if rec is None: return None - wav_path = rec.stop() - if not wav_path: - return None - return _transcribe_wav(wav_path, "voice transcription failed: %s") + return _transcribe_wav(wav_path, "voice transcription failed: %s") if wav_path else None # ── Continuous (VAD) API ───────────────────────────────────────────── @@ -406,11 +363,9 @@ def start_continuous( rec._max_recording_seconds = max_recording_seconds if cap_ok and max_recording_seconds > 0 else 0.0 _debug(f"start_continuous: begin (threshold={silence_threshold}, duration={silence_duration}s)") - - # CLI parity: beep *before* opening the stream — after stream.start() it - # triggers a CoreAudio conflict on macOS. + # CLI parity: beep *before* opening the stream — after stream.start() it triggers a CoreAudio + # conflict on macOS. _play_beep(frequency=880, count=1) - try: rec.start(on_silence_stop=_continuous_on_silence) except Exception as e: @@ -418,7 +373,6 @@ def start_continuous( _debug(f"start_continuous: rec.start raised {type(e).__name__}: {e}") _deactivate() raise - _safe_call(on_status, "listening") return True @@ -438,7 +392,7 @@ def stop_continuous(force_transcribe: bool = False) -> None: return _continuous_active = False rec = _continuous_recorder - on_transcript, on_status, on_silent_limit, on_stop_phrase = _callbacks() + callbacks = _callbacks() track_no_speech = force_transcribe and not _continuous_auto_restart _continuous_stopping = rec is not None _continuous_on_transcript = _continuous_on_status = None @@ -446,6 +400,7 @@ def stop_continuous(force_transcribe: bool = False) -> None: if not track_no_speech: _continuous_no_speech_count = 0 + on_transcript, on_status = callbacks[0], callbacks[1] if rec is not None: if force_transcribe and on_transcript: _safe_call(on_status, "transcribing") @@ -455,39 +410,37 @@ def stop_continuous(force_transcribe: bool = False) -> None: logger.warning("failed to stop recorder: %s", e) _safe_call(rec.cancel, warn="failed to cancel recorder: %s") wav_path = None - - def _transcribe_and_cleanup(): - transcript = _transcribe_wav(wav_path, "failed to stop/transcribe recorder: %s") if wav_path else None - - # With auto_restart=False the CLIENT drives the loop, so a stop - # phrase must fire the stop signal — discarding the transcript - # alone would leave the conversation running forever. - transcript, stop_phrase, stop_text = _detect_stop_phrase( - transcript, "stop_continuous", "ending voice chat" - ) - if stop_phrase: - _signal_halt(True, stop_text, on_stop_phrase, on_silent_limit) - if transcript: - _safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s") - - if track_no_speech: - held = _voice_activity_held() - with _continuous_lock: - should_halt, _ = _tally_silence( - bool(transcript) or stop_phrase, held, "stop_continuous" - ) - if should_halt: - _safe_call(on_silent_limit) - _finish_stop(on_status) - - threading.Thread(target=_transcribe_and_cleanup, daemon=True).start() + threading.Thread( + target=_finish_forced_stop, args=(wav_path, callbacks, track_no_speech), daemon=True + ).start() return - # cancel() (not stop()) discards buffered frames — the loop is over, we - # don't want to transcribe a half-captured turn. + # cancel() (not stop()) discards buffered frames — the loop is over, we don't want to + # transcribe a half-captured turn. _safe_call(rec.cancel, warn="failed to cancel recorder: %s") _finish_stop(on_status) +def _finish_forced_stop(wav_path: Optional[str], callbacks: tuple, track_no_speech: bool) -> None: + """Background tail of ``stop_continuous(force_transcribe=True)``: transcribe, deliver, tally.""" + on_transcript, on_status, on_silent_limit, on_stop_phrase = callbacks + # With auto_restart=False the CLIENT drives the loop, so a stop phrase must fire the stop + # signal — discarding the transcript alone would leave the conversation running forever. + transcript, stop_phrase, stop_text = _turn_transcript( + wav_path, "failed to stop/transcribe recorder: %s", "stop_continuous", "ending voice chat" + ) + if stop_phrase: + _signal_halt(True, stop_text, on_stop_phrase, on_silent_limit) + if transcript: + _safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s") + if track_no_speech: + held = _voice_activity_held() + with _continuous_lock: + should_halt, _ = _tally_silence(bool(transcript) or stop_phrase, held, "stop_continuous") + if should_halt: + _safe_call(on_silent_limit) + _finish_stop(on_status) + + def _finish_stop(on_status) -> None: """Clear the stopping flag, play the CLI-parity 660 Hz × 2 "stopped" cue, report idle.""" global _continuous_stopping @@ -512,59 +465,40 @@ def _continuous_on_silence() -> None: global _continuous_active, _continuous_no_speech_count _debug("_continuous_on_silence: fired") - with _continuous_lock: if not _continuous_active: _debug("_continuous_on_silence: loop inactive — abort") return rec = _continuous_recorder on_transcript, on_status, on_silent_limit, on_stop_phrase = _callbacks() - if rec is None: _debug("_continuous_on_silence: no recorder — abort") return _safe_call(on_status, "transcribing") - wav_path = rec.stop() - # Peak RMS tells at a glance whether the mic was too quiet for - # SILENCE_RMS_THRESHOLD (200) when stop() returns None despite the VAD firing. - peak_rms = getattr(rec, "_peak_rms", -1) - _debug(f"_continuous_on_silence: rec.stop -> {wav_path!r} (peak_rms={peak_rms})") - + # Peak RMS tells at a glance whether the mic was too quiet for SILENCE_RMS_THRESHOLD (200) + # when stop() returns None despite the VAD firing. + _debug(f"_continuous_on_silence: rec.stop -> {wav_path!r} (peak_rms={getattr(rec, '_peak_rms', -1)})") # CLI parity: double beep after the stream stops (safe from the CoreAudio conflict). _play_beep(frequency=660, count=2) - transcript = ( - _transcribe_wav(wav_path, "continuous transcription failed: %s", "_continuous_on_silence") - if wav_path else None + transcript, stop_phrase, stop_text = _turn_transcript( + wav_path, "continuous transcription failed: %s", "_continuous_on_silence", "ending loop", + debug_prefix="_continuous_on_silence", ) - - transcript, stop_phrase, stop_text = _detect_stop_phrase( - transcript, "_continuous_on_silence", "ending loop" - ) - # Held check runs outside the lock (the probe may call into the host surface). - _silence_held = transcript is None and not stop_phrase and _voice_activity_held() - + held = transcript is None and not stop_phrase and _voice_activity_held() with _continuous_lock: if not _continuous_active: - # User stopped us while we were transcribing — discard. _debug("_continuous_on_silence: stopped during transcribe — no restart") return - limit_hit, no_speech = _tally_silence( - bool(transcript) or stop_phrase, _silence_held, "_continuous_on_silence" - ) - should_halt = stop_phrase or limit_hit + limit_hit, no_speech = _tally_silence(bool(transcript) or stop_phrase, held, "_continuous_on_silence") if transcript: _safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s") - - if should_halt: - _debug( - "_continuous_on_silence: halting " - f"({'stop phrase' if stop_phrase else f'{no_speech} silent cycles'})" - ) + if stop_phrase or limit_hit: + _debug(f"_continuous_on_silence: halting ({'stop phrase' if stop_phrase else f'{no_speech} silent cycles'})") with _continuous_lock: _continuous_active = False _continuous_no_speech_count = 0 @@ -572,35 +506,37 @@ def _continuous_on_silence() -> None: _safe_call(rec.cancel) _safe_call(on_status, "idle") return + _rearm_after_turn(rec, on_status, no_speech) - # CLI parity: wait for in-flight TTS before re-arming the mic, then leave a - # small gap so the speaker tail isn't captured (otherwise the agent's spoken - # reply lands back in the mic and gets re-submitted). + +def _rearm_after_turn(rec: Any, on_status, no_speech: int) -> None: + """Wait out in-flight TTS, then restart capture (auto_restart) or stop (client-driven loop). + + CLI parity: the mic waits for TTS and then leaves a small gap so the speaker tail isn't + captured — otherwise the agent's spoken reply lands back in the mic and gets re-submitted. + """ if not _tts_playing.is_set(): _debug("_continuous_on_silence: waiting for TTS to finish") _tts_playing.wait(timeout=60) time.sleep(0.3) - with _continuous_lock: if not _continuous_active: _debug("_continuous_on_silence: stopped while waiting for TTS") return - - if _continuous_auto_restart: - _debug(f"_continuous_on_silence: restarting loop (no_speech={no_speech})") - _play_beep(frequency=880, count=1) - try: - rec.start(on_silence_stop=_continuous_on_silence) - except Exception as e: - logger.error("failed to restart continuous recording: %s", e) - _debug(f"_continuous_on_silence: restart raised {type(e).__name__}: {e}") - _deactivate(on_status) - return - - _safe_call(on_status, "listening") - else: + if not _continuous_auto_restart: _debug("_continuous_on_silence: auto_restart=False, stopping loop") _deactivate(on_status) + return + _debug(f"_continuous_on_silence: restarting loop (no_speech={no_speech})") + _play_beep(frequency=880, count=1) + try: + rec.start(on_silence_stop=_continuous_on_silence) + except Exception as e: + logger.error("failed to restart continuous recording: %s", e) + _debug(f"_continuous_on_silence: restart raised {type(e).__name__}: {e}") + _deactivate(on_status) + return + _safe_call(on_status, "listening") # ── TTS API ────────────────────────────────────────────────────────── @@ -620,17 +556,21 @@ _LEGACY_TTS_STRIP = [ ] -def _speak_text_streaming(text: str, stop_event: Optional[threading.Event] = None) -> bool: - """Speak ``text`` via the shared ``stream_tts_to_speaker`` pipeline; True on success. +def _speak_streaming(text: str, stop_event: Optional[threading.Event]) -> bool: + """Speak via the CLI's ``stream_tts_to_speaker`` pipeline when the configured provider has a + chunked streamer (audio starts on sentence one); False → caller uses the whole-file path. The full reply is fed as one delta + end-of-text sentinel and we block until the done event - fires — same blocking contract as the sync path, just earlier first audio. ``stop_event`` - is wired into the pipeline so external barge-in / stop paths can cut playback. + fires — same blocking contract as the sync path, just earlier first audio. ``stop_event`` is + wired into the pipeline so external barge-in / stop paths can cut playback. """ import queue - from tools.tts_tool import stream_tts_to_speaker + from tools.tts_streaming import resolve_streaming_provider + from tools.tts_tool import _load_tts_config, stream_tts_to_speaker + if resolve_streaming_provider(_load_tts_config()) is None: + return False text_queue: "queue.Queue" = queue.Queue() text_queue.put(text) text_queue.put(None) # end-of-text sentinel @@ -639,18 +579,62 @@ def _speak_text_streaming(text: str, stop_event: Optional[threading.Event] = Non return done_event.is_set() +def _speak_whole_file(text: str) -> None: + """Sync path: clean, synthesize to a temp MP3 via ``text_to_speech_tool``, play, unlink.""" + from tools.tts_tool import text_to_speech_tool + + # Shared cleaner (markdown, emoji, ⋗ blocks, verifier footer, units); the TTS tool owns + # provider request limits and long-form chunking. + try: + from tools.tts_text_normalize import prepare_spoken_text + tts_text = prepare_spoken_text(text, max_chars=None) + except Exception: + tts_text = text + for pattern, repl in _LEGACY_TTS_STRIP: + tts_text = pattern.sub(repl, tts_text) + tts_text = tts_text.strip() + if not tts_text: + return + + # Pre-chosen MP3 path so we can play MP3 even when text_to_speech_tool auto-converts to OGG + # for messaging platforms (afplay's OGG is flaky). + os.makedirs(os.path.join(tempfile.gettempdir(), "hermes_voice"), exist_ok=True) + mp3_path = os.path.join(tempfile.gettempdir(), "hermes_voice", f"tts_{time.strftime('%Y%m%d_%H%M%S')}.mp3") + _debug(f"speak_text: synthesizing {len(tts_text)} chars -> {mp3_path}") + raw_result = text_to_speech_tool(text=tts_text, output_path=mp3_path) + try: + tts_result = json.loads(raw_result) if isinstance(raw_result, str) else {} + except Exception: + tts_result = {} + + # The tool result is authoritative — long-form output may be several files. + play_paths = tts_result.get("file_paths") or [tts_result.get("file_path") or mp3_path] + played_any = False + for play_path in play_paths if tts_result.get("success") else []: + if os.path.isfile(play_path) and os.path.getsize(play_path) > 0: + _debug(f"speak_text: playing {play_path} ({os.path.getsize(play_path)} bytes)") + play_audio_file(play_path) + played_any = True + for path in set(play_paths + [mp3_path, mp3_path.rsplit(".", 1)[0] + ".ogg"]): + if os.path.isfile(path): + with contextlib.suppress(OSError): + os.unlink(path) + if not played_any: + _debug(f"speak_text: TTS tool produced no audio at {mp3_path}") + + def speak_text(text: str, stop_event: Optional[threading.Event] = None) -> None: """Synthesize ``text`` with the configured TTS provider and play it. While playback is in flight ``_tts_playing`` is cleared so the continuous loop waits before - re-arming the mic (otherwise the agent's reply feedback-loops through the microphone). + re-arming the mic (otherwise the agent's reply feedback-loops through the microphone). Live + capture is cancelled first — otherwise the user's turn tail + our first syllables both land + in the next recording window — and resumed afterwards so the user can answer without + pressing the record key. """ if not text or not text.strip(): return - # Cancel any live capture before opening the speakers — otherwise the user's - # turn tail + our first syllables both land in the next recording window. - # The loop re-arms after _tts_playing flips back (see _continuous_on_silence). paused_recording = False with _continuous_lock: if _continuous_active and getattr(_continuous_recorder, "is_recording", False): @@ -662,76 +646,23 @@ def speak_text(text: str, stop_event: Optional[threading.Event] = None) -> None: _tts_playing.clear() _debug(f"speak_text: TTS begin (paused_recording={paused_recording})") - try: - from tools.tts_tool import text_to_speech_tool + from tools.tts_tool import text_to_speech_tool # noqa: F401 (fail early, before streaming) - # One dispatcher: when the configured provider has a chunked streamer in - # tools.tts_streaming, route through the same stream_tts_to_speaker - # pipeline the CLI uses (audio starts on sentence one). Falls through to - # the whole-file path when no streamer resolves. + # One dispatcher: streaming when a chunked streamer resolves, else the whole-file path. try: - from tools.tts_streaming import resolve_streaming_provider - from tools.tts_tool import _load_tts_config - - if ( - resolve_streaming_provider(_load_tts_config()) is not None - and _speak_text_streaming(text, stop_event) - ): + if _speak_streaming(text, stop_event): return except Exception as e: _debug(f"speak_text: streaming dispatch unavailable ({e}); using sync path") - - # Shared cleaner (markdown, emoji, ⋗ blocks, verifier footer, units); - # the TTS tool owns provider request limits and long-form chunking. - try: - from tools.tts_text_normalize import prepare_spoken_text - tts_text = prepare_spoken_text(text, max_chars=None) - except Exception: - tts_text = text - for pattern, repl in _LEGACY_TTS_STRIP: - tts_text = pattern.sub(repl, tts_text) - tts_text = tts_text.strip() - if not tts_text: - return - - # Pre-chosen MP3 path so we can play MP3 even when text_to_speech_tool - # auto-converts to OGG for messaging platforms (afplay's OGG is flaky). - os.makedirs(os.path.join(tempfile.gettempdir(), "hermes_voice"), exist_ok=True) - mp3_path = os.path.join( - tempfile.gettempdir(), "hermes_voice", f"tts_{time.strftime('%Y%m%d_%H%M%S')}.mp3" - ) - - _debug(f"speak_text: synthesizing {len(tts_text)} chars -> {mp3_path}") - raw_result = text_to_speech_tool(text=tts_text, output_path=mp3_path) - try: - tts_result = json.loads(raw_result) if isinstance(raw_result, str) else {} - except Exception: - tts_result = {} - - # The tool result is authoritative — long-form output may be several files. - play_paths = tts_result.get("file_paths") or [tts_result.get("file_path") or mp3_path] - played_any = False - for play_path in play_paths if tts_result.get("success") else []: - if os.path.isfile(play_path) and os.path.getsize(play_path) > 0: - _debug(f"speak_text: playing {play_path} ({os.path.getsize(play_path)} bytes)") - play_audio_file(play_path) - played_any = True - for path in set(play_paths + [mp3_path, mp3_path.rsplit(".", 1)[0] + ".ogg"]): - if os.path.isfile(path): - with contextlib.suppress(OSError): - os.unlink(path) - if not played_any: - _debug(f"speak_text: TTS tool produced no audio at {mp3_path}") + _speak_whole_file(text) except Exception as e: logger.warning("Voice TTS playback failed: %s", e) _debug(f"speak_text raised {type(e).__name__}: {e}") finally: _tts_playing.set() _debug("speak_text: TTS done") - - # Re-arm the mic so the user can answer without pressing Ctrl+B. The - # delay lets afplay release the audio device before sounddevice re-opens. + # The delay lets afplay release the audio device before sounddevice re-opens. if paused_recording: time.sleep(0.3) with _continuous_lock: