fix(gateway): do not auto-TTS A2A replies

voice.auto_tts (flipped globally by Desktop "Read replies aloud") made
the runner synthesize a spoken reply for every A2A text task. The A2A
adapter has no native send_voice, so delivery fell back to the media
notice — the peer received "Couldn't deliver the audio attachment."
instead of the text the agent had already produced (#90103). Inbound A2A
is MessageType.TEXT; the base adapter's own auto-TTS gate keys on
MessageType.VOICE, but the runner's fallback branch
(voice_mode is None and adapter_auto_tts) had no platform gate at all.

Two aligned gates, both field-tested by the reporter:

- _should_send_voice_reply returns False for platform 'a2a' before any
  mode/config consultation — the text reply lands normally.
- _sync_voice_mode_state_to_adapter never pushes the global speak
  default onto the A2A adapter, so the adapter-side path cannot regress
  it either.

/voice on|tts|off scoped behavior is untouched (persisted per
platform:chat_id, still honored for human platforms). Tests pin the
pair the issue asked for: A2A + global auto-TTS skips, Telegram with
the same default still voices, and the sync-side skip.

Re-implemented from PR #90121 on the GatewayVoiceMixin split (author
credited). The Desktop-scoped read-aloud preference is a separate
design change, not folded in.

Fixes #90103
This commit is contained in:
mooserini
2026-09-24 18:44:06 -05:00
committed by brooklyn!
parent 55e27e5dbe
commit fa655b2980
2 changed files with 66 additions and 1 deletions

View File

@@ -122,7 +122,15 @@ class GatewayVoiceMixin:
except Exception:
auto_tts_default = False
if hasattr(adapter, "_auto_tts_default"):
adapter._auto_tts_default = auto_tts_default
# A2A never inherits the global speak default. The flag is written by
# Desktop "Read replies aloud" / voice.auto_tts and is meant for human
# chat surfaces; an agent peer's replies must stay text (its adapter has
# no native send_voice, so auto-TTS would synthesize an MP3 and then
# fail delivery) — /voice scoped modes never applied to A2A anyway.
if platform.value == "a2a":
adapter._auto_tts_default = False
else:
adapter._auto_tts_default = auto_tts_default
prefix = self._voice_key(platform, "", profile=getattr(adapter, "_owner_profile", None))
for chats, modes in chat_sets:
chats.clear()
@@ -296,6 +304,13 @@ class GatewayVoiceMixin:
— UNLESS streaming consumed the response (already_sent): then the runner must do it."""
if not response or response.startswith("Error:"):
return False
# A2A is agent-to-agent text. The adapter has no native send_voice, so
# global voice.auto_tts (Desktop "Read replies aloud") would synthesize
# an MP3 and then fail delivery with "Couldn't deliver the audio
# attachment." — the peer sees the failure instead of the text reply the
# agent already produced (#90103). Keep /voice scoped to human platforms.
if getattr(event.source.platform, "value", None) == "a2a":
return False
chat_id = event.source.chat_id
voice_mode = self._voice_mode.get(self._voice_key_for_source(event.source))
is_voice_input = event.message_type == MessageType.VOICE

View File

@@ -93,6 +93,56 @@ class TestAutoVoiceReplyFormat:
voice_event = _make_event(Platform.TELEGRAM, chat_id="123", message_type=MessageType.VOICE)
assert runner._should_send_voice_reply(voice_event, "hello", [], already_sent=True) is True
def test_should_send_voice_reply_a2a_ignores_global_auto_tts(self):
"""A2A text tasks must stay text even when voice.auto_tts is on.
Desktop Read-replies-aloud writes the global default. The runner used
that default for every adapter with no /voice mode, including A2A,
which cannot deliver native audio: the synthesized MP3 failed delivery
and the peer saw "Couldn't deliver the audio attachment." instead of
the agent's text reply (#90103).
"""
runner = _make_runner()
a2a = Platform("a2a")
adapter = _make_adapter(a2a)
adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
runner.adapters[a2a] = adapter
event = _make_event(a2a, chat_id="ctx-peer")
assert runner._should_send_voice_reply(event, "audit findings", []) is False
# The same global default still voices a human platform.
telegram_adapter = _make_adapter(Platform.TELEGRAM)
telegram_adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
runner.adapters[Platform.TELEGRAM] = telegram_adapter
telegram = _make_event(Platform.TELEGRAM, chat_id="999")
assert runner._should_send_voice_reply(telegram, "hello", []) is True
def test_sync_voice_mode_state_never_inherits_global_auto_tts_for_a2a(self):
"""The adapter-side default must match the runner-side skip (#90103).
voice.auto_tts is synced onto every adapter at connect; for A2A the
base adapter's own auto-TTS gate would otherwise read a speak default
the platform cannot honor.
"""
runner = _make_runner()
a2a = Platform("a2a")
a2a_adapter = _make_adapter(a2a)
a2a_adapter._auto_tts_disabled_chats = set()
a2a_adapter._auto_tts_enabled_chats = set()
telegram_adapter = _make_adapter(Platform.TELEGRAM)
telegram_adapter._auto_tts_disabled_chats = set()
telegram_adapter._auto_tts_enabled_chats = set()
with patch("hermes_cli.config.load_config", return_value={"voice": {"auto_tts": True}}):
runner._sync_voice_mode_state_to_adapter(a2a_adapter)
runner._sync_voice_mode_state_to_adapter(telegram_adapter)
assert a2a_adapter._auto_tts_default is False
assert telegram_adapter._auto_tts_default is True
def _make_runner() -> GatewayRunner:
with patch("gateway.run.GatewayRunner._load_voice_modes", return_value={}):
runner = GatewayRunner.__new__(GatewayRunner)