Files
hermes-agent/hermes_cli/voice.py

905 lines
37 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Process-wide voice recording + TTS API for the TUI gateway."""
from __future__ import annotations
import json
import logging
import os
import re
import sys
import threading
from typing import Any, Callable, Optional
# Modifier aliases mirrored from the TUI parser (``ui-tui/src/lib/platform.ts``)
# ``_MOD_ALIASES`` table — the contract that removes the cross-runtime
# mismatch Copilot flagged in round-9 on #19835.
#
# ``super``/``win``/``windows`` are intentionally absent: prompt_toolkit
# has no super/meta modifier for the Cmd key, so those spellings are
# TUI-only. The normalizer below returns the documented default
# (``c-b``) for them — a silent fallback was preferred to a hard
# startup crash (Copilot round-11). The CLI binding site
# (``_register_voice_handler`` in cli.py) logs a warning when that
# fallback fires so users see why their TUI-only shortcut isn't
# bound in the classic CLI.
_VOICE_MOD_ALIASES = {
"ctrl": "c-",
"control": "c-",
"alt": "a-",
"option": "a-",
"opt": "a-",
}
# Named keys prompt_toolkit accepts in ``c-<name>`` / ``a-<name>`` form.
# Aliases collapse to prompt_toolkit's canonical spelling so the same
# config value binds identically in both runtimes (Copilot round-10 on
# #19835).
_VOICE_NAMED_KEYS = {
"space": "space",
"spc": "space",
"enter": "enter",
"return": "enter",
"ret": "enter",
"tab": "tab",
"escape": "escape",
"esc": "escape",
"backspace": "backspace",
"bs": "backspace",
"delete": "delete",
"del": "delete",
}
# ``useInputHandlers()`` intercepts these before the voice check runs,
# so a binding like ``ctrl+c`` (interrupt), ``ctrl+d`` (quit), or
# ``ctrl+l`` (clear screen) would be advertised in /voice status but
# never fire push-to-talk — the same blocklist the TUI parser uses.
_VOICE_RESERVED_CTRL_CHARS = frozenset({"c", "d", "l"})
# On macOS the classic CLI's prompt_toolkit bindings for copy / exit /
# clear also claim ``a-c`` / ``a-d`` / ``a-l`` via the action-modifier
# lookup, and hermes-ink reports Alt as ``key.meta`` on many terminals.
# Mirror the TUI parser's darwin-only reservation so ``option+c`` etc.
# don't bind Alt+C in the CLI while the TUI silently falls back to
# Ctrl+B (Copilot round-14 on #19835).
_VOICE_RESERVED_ALT_CHARS_MAC = frozenset({"c", "d", "l"})
_DEFAULT_PT_KEY = "c-b"
def voice_record_key_from_config(cfg: Any) -> Any:
"""Shape-safe ``cfg.voice.record_key`` lookup.
``load_config()`` deep-merges raw YAML and preserves scalar overrides, so a hand-edited ``voice:
true`` / ``voice: cmd+b`` leaves ``cfg["voice"]`` as a bool/str instead of a dict, and the naive
``.get("voice", {}).get("record_key")`` chain raises AttributeError before voice can even start
(Copilot round-11 on #19835).
"""
voice = cfg.get("voice") if isinstance(cfg, dict) else None
return voice.get("record_key") if isinstance(voice, dict) else None
def normalize_voice_record_key_for_prompt_toolkit(raw: Any) -> str:
"""Coerce ``voice.record_key`` into prompt_toolkit's ``c-x`` / ``a-x`` format.
Mirrors the TUI parser contract (``ui-tui/src/lib/platform.ts``) so one config value binds the
same shortcut in both runtimes:
* non-string / empty / typo'd / bare-char / multi-modifier / reserved ``ctrl+c|d|l`` →
documented default ``c-b`` * single-char keys: ``ctrl+o`` → ``c-o`` * named keys: ``ctrl+space``
→ ``c-space`` (aliases collapse: ``ctrl+return`` → ``c-enter``) * ``super`` / ``win`` /
``windows`` → ``c-b`` (TUI-only modifiers — prompt_toolkit has no super mod; the CLI binding
site is expected to warn when this fallback fires so users see the cross-runtime split, Copilot
round-11 on #19835)
"""
if not isinstance(raw, str):
return _DEFAULT_PT_KEY
parts = [p.strip() for p in raw.strip().lower().split("+") if p.strip()]
# Exactly ``modifier+key``. Multi-modifier chords like ``ctrl+alt+r`` bind
# different shortcuts in prompt_toolkit (a-c-r form) and hermes-ink rejects
# them; a bare char / named key (no modifier) is refused by the TUI parser.
# Both collapse to the documented default so the runtimes agree.
if len(parts) != 2:
return _DEFAULT_PT_KEY
modifier_token, key_token = parts
# ``super`` / ``win`` / ``windows`` are TUI-only (prompt_toolkit has
# no super modifier, so ``@kb.add(super+b)`` crashes the CLI at
# startup). Fall back to the documented default here; the CLI
# binding site is expected to log a warning when the configured
# value is one of these spellings so users know the TUI+CLI
# runtimes diverge on that shortcut (Copilot round-11 on #19835).
if modifier_token in {"super", "win", "windows"}:
return _DEFAULT_PT_KEY
normalized_mod = _VOICE_MOD_ALIASES.get(modifier_token)
if not normalized_mod:
return _DEFAULT_PT_KEY
# Single-char key: reject reserved-ctrl chords that the TUI would
# also block at parse time, plus the mac-only alt reservation.
if len(key_token) == 1:
reserved = (
_VOICE_RESERVED_CTRL_CHARS if normalized_mod == "c-"
else _VOICE_RESERVED_ALT_CHARS_MAC if sys.platform == "darwin"
else frozenset()
)
return _DEFAULT_PT_KEY if key_token in reserved else f"{normalized_mod}{key_token}"
# Multi-char key token must be a known named key; typos like
# ``ctrl+spcae`` fall back to the default rather than being passed
# through as ``c-spcae`` (which prompt_toolkit would reject).
named = _VOICE_NAMED_KEYS.get(key_token)
return f"{normalized_mod}{named}" if named else _DEFAULT_PT_KEY
def pt_key_to_sequence(pt_key: str) -> tuple[str, ...]:
"""Convert a prompt_toolkit key specifier (e.g. 'c-b' or 'a-v') to a sequence tuple."""
if isinstance(pt_key, str) and pt_key.startswith("a-"):
return ("escape", pt_key[2:])
return (pt_key,)
def format_voice_record_key_for_status(raw: Any) -> str:
"""Render ``voice.record_key`` for ``/voice status`` in CLI-friendly form.
Mirrors the TUI's ``formatVoiceRecordKey``: returns ``Ctrl+B`` / ``Alt+Space`` / ``Ctrl+Enter``.
Malformed configs surface as the documented default so status never advertises a shortcut that
won't bind (Copilot round-10 on #19835).
"""
# The normalizer only ever yields ``c-<key>`` / ``a-<key>`` (or the default ``c-b``).
normalized = normalize_voice_record_key_for_prompt_toolkit(raw)
prefix = "Alt+" if normalized.startswith("a-") else "Ctrl+"
key = normalized[2:]
return prefix + key[0].upper() + key[1:]
from tools.voice_mode import (
create_audio_recorder,
is_voice_stop_phrase,
is_whisper_hallucination,
play_audio_file,
transcribe_recording,
)
logger = logging.getLogger(__name__)
def _debug(msg: str) -> None:
"""Emit a debug breadcrumb when HERMES_VOICE_DEBUG=1.
Goes to stderr so the TUI gateway wraps it as a gateway.stderr event, which
createGatewayEventHandler shows as an Activity line — exactly what we need to diagnose "why
didn't the loop auto-restart?" in the user's real terminal without shipping a separate debug
RPC.
Any OSError / BrokenPipeError is swallowed because this fires from background threads (silence
callback, TTS daemon, beep) where a broken stderr pipe must not kill the whole gateway — the
main command pipe (stdin+stdout) is what actually matters.
"""
if os.environ.get("HERMES_VOICE_DEBUG", "").strip() != "1":
return
try:
print(f"[voice] {msg}", file=sys.stderr, flush=True)
except (BrokenPipeError, OSError):
pass
def _beeps_enabled() -> bool:
"""CLI parity: voice.beep_enabled in config.yaml (default True)."""
try:
from hermes_cli.config import load_config
from utils import is_truthy_value
voice_cfg = load_config().get("voice", {})
if isinstance(voice_cfg, dict):
# is_truthy_value handles quoted YAML strings like "false"
# which bool() would misread as True (#49883).
return is_truthy_value(voice_cfg.get("beep_enabled", True), default=True)
except Exception:
pass
return True
def _play_beep(frequency: int, count: int = 1) -> None:
"""Audible cue matching cli.py's record/stop beeps.
880 Hz single-beep on start (cli.py:_voice_start_recording line 7532), 660 Hz double-beep on
stop (cli.py:_voice_stop_and_transcribe line 7585). Best-effort — sounddevice failures are
silently swallowed so the voice loop never breaks because a speaker was unavailable.
"""
if not _beeps_enabled():
return
try:
from tools.voice_mode import play_beep
play_beep(frequency=frequency, count=count)
except Exception as e:
_debug(f"beep {frequency}Hz failed: {e}")
def _safe_call(cb: Optional[Callable], *args: Any, warn: Optional[str] = None) -> None:
"""Invoke an optional consumer callback, swallowing its exceptions.
``warn`` is a ``logger.warning`` format with one ``%s`` slot for the exception; without it
failures are silently ignored (status/limit callbacks are fire-and-forget).
"""
if not cb:
return
try:
cb(*args)
except Exception as e:
if warn:
logger.warning(warn, e)
def _transcribe_wav(wav_path: str, fail_msg: str, debug_prefix: Optional[str] = None) -> Optional[str]:
"""Transcribe ``wav_path``, unlink it, and return the cleaned transcript (or None).
transcribe_recording returns {"success": bool, "transcript": str, "error": str?} — NOT
{"text": str}. Using the wrong key silently produced empty transcripts even when Groq/local
STT returned fine, which masqueraded as "not hearing the user" to the caller. Empty text and
Whisper hallucinations are dropped; failures are logged with ``fail_msg``.
"""
try:
result = transcribe_recording(wav_path)
success = bool(result.get("success"))
text = (result.get("transcript") or "").strip()
if debug_prefix:
_debug(
f"{debug_prefix}: transcribe -> success={success} "
f"text={text!r} err={result.get('error')!r}"
)
if success and text and not is_whisper_hallucination(text):
return text
except Exception as e:
logger.warning(fail_msg, e)
if debug_prefix:
_debug(f"{debug_prefix}: transcribe raised {type(e).__name__}: {e}")
finally:
try:
if os.path.isfile(wav_path):
os.unlink(wav_path)
except Exception:
pass
return None
def _deactivate(on_status: Optional[Callable[[str], None]] = None) -> None:
"""Mark the continuous loop inactive and (optionally) report ``"idle"``."""
global _continuous_active
with _continuous_lock:
_continuous_active = False
_safe_call(on_status, "idle")
# ── Push-to-talk state ───────────────────────────────────────────────
_recorder = None
_recorder_lock = threading.Lock()
# ── Continuous (VAD) state ───────────────────────────────────────────
_continuous_lock = threading.Lock()
_continuous_active = False
_continuous_stopping = False
_continuous_auto_restart: bool = True
_continuous_recorder: Any = None
# ── TTS-vs-STT feedback guard ────────────────────────────────────────
# When TTS plays the agent reply over the speakers, the live microphone
# picks it up and transcribes the agent's own voice as user input — an
# infinite loop the agent happily joins ("Ha, looks like we're in a loop").
# This Event mirrors cli.py:_voice_tts_done: cleared while speak_text is
# playing, set while silent. _continuous_on_silence waits on it before
# re-arming the recorder, and speak_text itself cancels any live capture
# before starting playback so the tail of the previous utterance doesn't
# leak into the mic.
_tts_playing = threading.Event()
_tts_playing.set() # initially "not playing"
# ── Silence-count hold (agent busy) ──────────────────────────────────
# While the agent is mid-turn (thinking / tool-calling, possibly for
# minutes) or TTS is playing, the user is CORRECTLY silent — those cycles
# must not count toward the no-speech limit or a long tool run ends the
# voice chat under the user (#silence-must-not-end-the-chat). The host
# surface (tui_gateway) registers a probe that reports "agent busy";
# TTS-playing is already tracked via _tts_playing above.
_voice_busy_probe: Optional[Callable[[], bool]] = None
def set_voice_busy_probe(probe: Optional[Callable[[], bool]]) -> None:
"""Register a callable that returns True while the agent is mid-turn.
Called by the hosting surface (tui_gateway registers one that checks every session's ``running``
flag). ``None`` clears it. The probe must be cheap and thread-safe — it runs on the silence-
callback thread.
"""
global _voice_busy_probe
_voice_busy_probe = probe
def _voice_activity_held() -> bool:
"""True while silent cycles must NOT count toward the no-speech limit.
Held when TTS is playing (the user is listening) or when the registered busy probe reports the
agent mid-turn (the user is waiting). Fail-open to "not held" so a broken probe can never make
the voice chat immortal.
"""
if not _tts_playing.is_set():
return True
probe = _voice_busy_probe
if probe is None:
return False
try:
return bool(probe())
except Exception:
return False
_continuous_on_transcript: Optional[Callable[[str], None]] = None
_continuous_on_status: Optional[Callable[[str], None]] = None
_continuous_on_silent_limit: Optional[Callable[[], None]] = None
# Explicit user-intent stop signal: fired when the user SAYS a bare stop
# phrase ("stop"). Distinct from on_silent_limit (a timeout) so consumers
# (TUI, desktop) can end the conversation like a manual stop instead of
# reporting "no speech detected". When unset, on_silent_limit fires as a
# fallback so older callers still turn voice off.
_continuous_on_stop_phrase: Optional[Callable[[str], None]] = None
_continuous_no_speech_count = 0
_CONTINUOUS_NO_SPEECH_LIMIT = 3
# ── Push-to-talk API ─────────────────────────────────────────────────
def start_recording() -> None:
"""Begin capturing from the default input device (push-to-talk)."""
global _recorder
with _recorder_lock:
if _recorder is not None and getattr(_recorder, "is_recording", False):
return
rec = create_audio_recorder()
rec.start()
_recorder = rec
def stop_and_transcribe() -> Optional[str]:
"""Stop the active push-to-talk recording, transcribe, return text."""
global _recorder
with _recorder_lock:
rec = _recorder
_recorder = None
if rec is None:
return None
wav_path = rec.stop()
if not wav_path:
return None
return _transcribe_wav(wav_path, "voice transcription failed: %s")
# ── Continuous (VAD) API ─────────────────────────────────────────────
def start_continuous(
on_transcript: Callable[[str], None],
on_status: Optional[Callable[[str], None]] = None,
on_silent_limit: Optional[Callable[[], None]] = None,
silence_threshold: int = 200,
silence_duration: float = 3.0,
auto_restart: bool = True,
max_recording_seconds: float = 0.0,
on_stop_phrase: Optional[Callable[[str], None]] = None,
) -> bool:
"""Start a VAD-driven continuous recording loop.
``max_recording_seconds`` is the hard cap on a single recording's length
(``voice.max_recording_seconds``); any non-positive or non-numeric value disables the cap,
preserving the previous unbounded behaviour.
``on_stop_phrase`` is called with the (stripped) transcript when the user utters a bare voice
stop phrase (``voice.stop_phrases``, default "stop"). The loop halts first, so the consumer only
needs to reflect "voice off" — exactly like the user pressing the manual stop control.
"""
global _continuous_active, _continuous_recorder, _continuous_auto_restart
global _continuous_on_transcript, _continuous_on_status, _continuous_on_silent_limit
global _continuous_on_stop_phrase
global _continuous_no_speech_count
with _continuous_lock:
if _continuous_active:
_debug("start_continuous: already active — no-op")
return True
if _continuous_stopping:
_debug("start_continuous: stop/transcribe in progress — busy")
return False
_continuous_active = True
_continuous_auto_restart = auto_restart
_continuous_on_transcript = on_transcript
_continuous_on_status = on_status
_continuous_on_silent_limit = on_silent_limit
_continuous_on_stop_phrase = on_stop_phrase
if auto_restart:
_continuous_no_speech_count = 0
if _continuous_recorder is None:
_continuous_recorder = create_audio_recorder()
rec = _continuous_recorder
rec._silence_threshold = silence_threshold
rec._silence_duration = silence_duration
# Same numeric-with-bool-excluded guard as the CLI wiring in
# cli.py:_voice_start_recording — <= 0 (or garbage) disables the cap.
rec._max_recording_seconds = (
max_recording_seconds
if isinstance(max_recording_seconds, (int, float))
and not isinstance(max_recording_seconds, bool)
and max_recording_seconds > 0
else 0.0
)
_debug(
f"start_continuous: begin (threshold={silence_threshold}, duration={silence_duration}s)"
)
# CLI parity: single 880 Hz beep *before* opening the stream — placing
# the beep after stream.start() on macOS triggers a CoreAudio conflict
# (cli.py:7528 comment).
_play_beep(frequency=880, count=1)
try:
rec.start(on_silence_stop=_continuous_on_silence)
except Exception as e:
logger.error("failed to start continuous recording: %s", e)
_debug(f"start_continuous: rec.start raised {type(e).__name__}: {e}")
_deactivate()
raise
_safe_call(on_status, "listening")
return True
def stop_continuous(force_transcribe: bool = False) -> None:
"""Stop the active continuous loop and release the microphone.
Idempotent — calling while not active is a no-op. If ``force_transcribe`` is True, the recorder
stops synchronously, then transcription/cleanup runs on a background thread before reporting
``"idle"``. Otherwise the buffer is discarded.
"""
global _continuous_active, _continuous_on_transcript, _continuous_stopping
global _continuous_on_status, _continuous_on_silent_limit
global _continuous_on_stop_phrase
global _continuous_recorder, _continuous_no_speech_count
with _continuous_lock:
if not _continuous_active:
return
_continuous_active = False
rec = _continuous_recorder
on_status = _continuous_on_status
on_transcript = _continuous_on_transcript
on_silent_limit = _continuous_on_silent_limit
on_stop_phrase = _continuous_on_stop_phrase
auto_restart = _continuous_auto_restart
track_no_speech = force_transcribe and not auto_restart
_continuous_stopping = rec is not None
_continuous_on_transcript = None
_continuous_on_status = None
_continuous_on_silent_limit = None
_continuous_on_stop_phrase = None
if not track_no_speech:
_continuous_no_speech_count = 0
if rec is not None:
if force_transcribe and on_transcript:
_safe_call(on_status, "transcribing")
try:
wav_path = rec.stop()
except Exception as e:
logger.warning("failed to stop recorder: %s", e)
try:
rec.cancel()
except Exception as cancel_error:
logger.warning("failed to cancel recorder: %s", cancel_error)
wav_path = None
def _transcribe_and_cleanup():
global _continuous_no_speech_count, _continuous_stopping
transcript: Optional[str] = None
should_halt = False
if wav_path:
transcript = _transcribe_wav(wav_path, "failed to stop/transcribe recorder: %s")
stop_phrase = bool(transcript and is_voice_stop_phrase(transcript))
if stop_phrase:
# Bare stop phrase — explicit user intent to end the
# voice chat. Never sent to the agent; fire the
# dedicated signal so the consumer (TUI / desktop)
# ends the conversation instead of silently re-arming
# the next capture (with auto_restart=False the CLIENT
# drives the loop, so discarding the transcript alone
# would leave the conversation running forever).
_debug(
f"stop_continuous: stop phrase {transcript!r} — ending voice chat"
)
stop_text = transcript or ""
transcript = None
if on_stop_phrase is not None:
_safe_call(on_stop_phrase, stop_text)
else:
_safe_call(on_silent_limit)
if transcript:
_safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s")
if track_no_speech:
held = _voice_activity_held()
with _continuous_lock:
if transcript or stop_phrase:
_continuous_no_speech_count = 0
elif held:
# Agent busy / TTS playing — the user is
# correctly silent; don't count the cycle.
_debug(
"stop_continuous: silent cycle ignored "
"(agent busy or TTS playing)"
)
else:
_continuous_no_speech_count += 1
should_halt = (
_continuous_no_speech_count
>= _CONTINUOUS_NO_SPEECH_LIMIT
)
if should_halt:
_continuous_no_speech_count = 0
if should_halt:
_safe_call(on_silent_limit)
_play_beep(frequency=660, count=2)
with _continuous_lock:
_continuous_stopping = False
_safe_call(on_status, "idle")
threading.Thread(target=_transcribe_and_cleanup, daemon=True).start()
return
else:
try:
# cancel() (not stop()) discards buffered frames — the loop
# is over, we don't want to transcribe a half-captured turn.
rec.cancel()
except Exception as e:
logger.warning("failed to cancel recorder: %s", e)
with _continuous_lock:
_continuous_stopping = False
# Audible "recording stopped" cue (CLI parity: same 660 Hz × 2 the
# silence-auto-stop path plays).
_play_beep(frequency=660, count=2)
_safe_call(on_status, "idle")
def is_continuous_active() -> bool:
"""Whether a continuous voice loop is currently running."""
with _continuous_lock:
return _continuous_active
def _continuous_on_silence() -> None:
"""AudioRecorder silence callback — runs in a daemon thread.
Stops the current capture, transcribes, delivers text via ``on_transcript``, and — if the
loop is still active — starts the next capture. Three consecutive silent cycles end the
loop.
"""
global _continuous_active, _continuous_no_speech_count
_debug("_continuous_on_silence: fired")
with _continuous_lock:
if not _continuous_active:
_debug("_continuous_on_silence: loop inactive — abort")
return
rec = _continuous_recorder
on_transcript = _continuous_on_transcript
on_status = _continuous_on_status
on_silent_limit = _continuous_on_silent_limit
on_stop_phrase = _continuous_on_stop_phrase
if rec is None:
_debug("_continuous_on_silence: no recorder — abort")
return
_safe_call(on_status, "transcribing")
wav_path = rec.stop()
# Peak RMS is the critical diagnostic when stop() returns None despite
# the VAD firing — tells us at a glance whether the mic was too quiet
# for SILENCE_RMS_THRESHOLD (200) or the VAD + peak checks disagree.
peak_rms = getattr(rec, "_peak_rms", -1)
_debug(
f"_continuous_on_silence: rec.stop -> {wav_path!r} (peak_rms={peak_rms})"
)
# CLI parity: double 660 Hz beep after the stream stops (safe from the
# CoreAudio conflict that blocks pre-start beeps).
_play_beep(frequency=660, count=2)
transcript: Optional[str] = None
if wav_path:
transcript = _transcribe_wav(
wav_path, "continuous transcription failed: %s", "_continuous_on_silence"
)
stop_phrase = bool(transcript and is_voice_stop_phrase(transcript))
stop_text = (transcript or "") if stop_phrase else ""
if stop_phrase:
# User said a bare stop phrase ("stop") — end the voice chat.
# Not delivered to the agent; the loop halts and the explicit
# on_stop_phrase signal (fallback: on_silent_limit) tells every UI
# (TUI, desktop) to end the conversation like a manual stop.
_debug(f"_continuous_on_silence: stop phrase {transcript!r} — ending loop")
transcript = None
# Silent cycle while the agent is mid-turn or TTS is playing: the user
# is CORRECTLY quiet (waiting/listening), so the cycle must not count
# toward the no-speech limit — a multi-minute tool run would otherwise
# end the voice chat under the user. Checked outside the lock (probe
# may call into the host surface).
_silence_held = (transcript is None and not stop_phrase
and _voice_activity_held())
with _continuous_lock:
if not _continuous_active:
# User stopped us while we were transcribing — discard.
_debug("_continuous_on_silence: stopped during transcribe — no restart")
return
if transcript:
_continuous_no_speech_count = 0
elif _silence_held:
_debug(
"_continuous_on_silence: silent cycle ignored "
"(agent busy or TTS playing)"
)
elif not stop_phrase:
_continuous_no_speech_count += 1
should_halt = stop_phrase or (
_continuous_no_speech_count >= _CONTINUOUS_NO_SPEECH_LIMIT
)
no_speech = _continuous_no_speech_count
if transcript:
_safe_call(on_transcript, transcript, warn="on_transcript callback raised: %s")
if should_halt:
_debug(
"_continuous_on_silence: halting "
f"({'stop phrase' if stop_phrase else f'{no_speech} silent cycles'})"
)
with _continuous_lock:
_continuous_active = False
_continuous_no_speech_count = 0
if stop_phrase and on_stop_phrase is not None:
# Explicit user-intent stop — distinct from the no-speech timeout
# so consumers can report "voice chat ended" instead of "no
# speech detected".
_safe_call(on_stop_phrase, stop_text)
else:
_safe_call(on_silent_limit)
_safe_call(rec.cancel)
_safe_call(on_status, "idle")
return
# CLI parity (cli.py:10619-10621): wait for any in-flight TTS to
# finish before re-arming the mic, then leave a small gap to avoid
# catching the tail of the speaker output. Without this the voice
# loop becomes a feedback loop — the agent's spoken reply lands
# back in the mic and gets re-submitted.
if not _tts_playing.is_set():
_debug("_continuous_on_silence: waiting for TTS to finish")
_tts_playing.wait(timeout=60)
import time as _time
_time.sleep(0.3)
# User may have stopped the loop during the wait.
with _continuous_lock:
if not _continuous_active:
_debug("_continuous_on_silence: stopped while waiting for TTS")
return
if _continuous_auto_restart:
# Restart for the next turn.
_debug(f"_continuous_on_silence: restarting loop (no_speech={no_speech})")
_play_beep(frequency=880, count=1)
try:
rec.start(on_silence_stop=_continuous_on_silence)
except Exception as e:
logger.error("failed to restart continuous recording: %s", e)
_debug(f"_continuous_on_silence: restart raised {type(e).__name__}: {e}")
_deactivate(on_status)
return
_safe_call(on_status, "listening")
else:
# Do not auto-restart. Clean up state and notify idle.
_debug("_continuous_on_silence: auto_restart=False, stopping loop")
_deactivate(on_status)
# ── TTS API ──────────────────────────────────────────────────────────
# Legacy markdown stripper used only when tools.tts_text_normalize is unavailable.
_LEGACY_TTS_STRIP = [
(re.compile(r'```[\s\S]*?```'), ' '), # fenced code blocks
(re.compile(r'\[([^\]]+)\]\([^)]+\)'), r'\1'), # [text](url) → text
(re.compile(r'https?://\S+'), ''), # bare URLs
(re.compile(r'\*\*(.+?)\*\*'), r'\1'), # bold
(re.compile(r'\*(.+?)\*'), r'\1'), # italic
(re.compile(r'`(.+?)`'), r'\1'), # inline code
(re.compile(r'^#+\s*', re.MULTILINE), ''), # headers
(re.compile(r'^\s*[-*]\s+', re.MULTILINE), ''), # list bullets
(re.compile(r'---+'), ''), # horizontal rules
(re.compile(r'\n{3,}'), '\n\n'), # excess newlines
]
def _speak_text_streaming(text: str, stop_event: Optional[threading.Event] = None) -> bool:
"""Speak ``text`` via the generic streaming dispatcher; True on success.
Bridges the one-shot ``speak_text`` contract onto the shared ``stream_tts_to_speaker`` pipeline
(tools.tts_tool): the full reply is fed as a single delta + end-of-text sentinel, and we block
until the pipeline's done event fires — same blocking semantics the sync path has, so callers
(and the mic re-arm logic in ``speak_text``) see no behavioral difference beyond earlier first
audio.
``stop_event`` (optional) is wired straight into the pipeline so external barge-in / stop paths
can cut streaming playback — without it the pipeline's stop event was private and speech over
this path was uninterruptible (the desktop/TUI fallback-speak hole).
"""
import queue as _queue
import threading as _threading
from tools.tts_tool import stream_tts_to_speaker
text_queue: "_queue.Queue" = _queue.Queue()
text_queue.put(text)
text_queue.put(None) # end-of-text sentinel
if stop_event is None:
stop_event = _threading.Event()
done_event = _threading.Event()
stream_tts_to_speaker(text_queue, stop_event, done_event)
return done_event.is_set()
def speak_text(text: str, stop_event: Optional[threading.Event] = None) -> None:
"""Synthesize ``text`` with the configured TTS provider and play it.
While playback is in flight the module-level _tts_playing Event is cleared so the continuous-
recording loop knows to wait before re-arming the mic (otherwise the agent's spoken reply
feedback-loops through the microphone and the agent ends up replying to itself).
"""
if not text or not text.strip():
return
import tempfile
import time
# Cancel any live capture before we open the speakers — otherwise the
# last ~200ms of the user's turn tail + the first syllables of our TTS
# both end up in the next recording window. The continuous loop will
# re-arm itself after _tts_playing flips back (see _continuous_on_silence).
paused_recording = False
with _continuous_lock:
if (
_continuous_active
and _continuous_recorder is not None
and getattr(_continuous_recorder, "is_recording", False)
):
try:
_continuous_recorder.cancel()
paused_recording = True
except Exception as e:
logger.warning("failed to pause recorder for TTS: %s", e)
_tts_playing.clear()
_debug(f"speak_text: TTS begin (paused_recording={paused_recording})")
try:
from tools.tts_tool import text_to_speech_tool
# One dispatcher, zero parallel streaming implementations (#58930):
# when the configured provider has a chunked streamer registered in
# tools.tts_streaming, route the whole reply through the same
# stream_tts_to_speaker pipeline the CLI voice mode uses — audio
# starts on sentence one instead of after full synthesis. Falls
# through to the legacy whole-file path when no streamer resolves.
try:
from tools.tts_streaming import resolve_streaming_provider
from tools.tts_tool import _load_tts_config
if (
resolve_streaming_provider(_load_tts_config()) is not None
and _speak_text_streaming(text, stop_event)
):
return
except Exception as e:
_debug(f"speak_text: streaming dispatch unavailable ({e}); using sync path")
# Shared cleaner (tools/tts_text_normalize): markdown, emoji,
# ⋗ blocks, verifier footer, units, newline flattening.
# The TTS tool owns provider request limits and long-form chunking.
try:
from tools.tts_text_normalize import prepare_spoken_text
tts_text = prepare_spoken_text(text, max_chars=None)
except Exception:
# Legacy fallback pipeline — keep speak_text best-effort.
tts_text = text
for pattern, repl in _LEGACY_TTS_STRIP:
tts_text = pattern.sub(repl, tts_text)
tts_text = tts_text.strip()
if not tts_text:
return
# MP3 output path, pre-chosen so we can play the MP3 directly even
# when text_to_speech_tool auto-converts to OGG for messaging
# platforms. afplay's OGG support is flaky, MP3 always works.
os.makedirs(os.path.join(tempfile.gettempdir(), "hermes_voice"), exist_ok=True)
mp3_path = os.path.join(
tempfile.gettempdir(),
"hermes_voice",
f"tts_{time.strftime('%Y%m%d_%H%M%S')}.mp3",
)
_debug(f"speak_text: synthesizing {len(tts_text)} chars -> {mp3_path}")
raw_result = text_to_speech_tool(text=tts_text, output_path=mp3_path)
try:
tts_result = json.loads(raw_result) if isinstance(raw_result, str) else {}
except Exception:
tts_result = {}
# The tool result is authoritative — it may return multiple files
# for long-form chunked output. Play each in order.
play_paths = tts_result.get("file_paths") or [
tts_result.get("file_path") or mp3_path
]
played_any = False
for play_path in play_paths if tts_result.get("success") else []:
if os.path.isfile(play_path) and os.path.getsize(play_path) > 0:
_debug(
f"speak_text: playing {play_path} "
f"({os.path.getsize(play_path)} bytes)"
)
play_audio_file(play_path)
played_any = True
for path in set(play_paths + [mp3_path, mp3_path.rsplit(".", 1)[0] + ".ogg"]):
if os.path.isfile(path):
try:
os.unlink(path)
except OSError:
pass
if not played_any:
_debug(f"speak_text: TTS tool produced no audio at {mp3_path}")
except Exception as e:
logger.warning("Voice TTS playback failed: %s", e)
_debug(f"speak_text raised {type(e).__name__}: {e}")
finally:
_tts_playing.set()
_debug("speak_text: TTS done")
# Re-arm the mic so the user can answer without pressing Ctrl+B.
# Small delay lets the OS flush speaker output and afplay fully
# release the audio device before sounddevice re-opens the input.
if paused_recording:
time.sleep(0.3)
with _continuous_lock:
if _continuous_active and _continuous_recorder is not None:
try:
_continuous_recorder.start(
on_silence_stop=_continuous_on_silence
)
_debug("speak_text: recording resumed after TTS")
except Exception as e:
logger.warning(
"failed to resume recorder after TTS: %s", e
)