diff --git a/gateway/run_voice.py b/gateway/run_voice.py index 06a8df1137..b5167ef5d8 100644 --- a/gateway/run_voice.py +++ b/gateway/run_voice.py @@ -122,7 +122,15 @@ class GatewayVoiceMixin: except Exception: auto_tts_default = False if hasattr(adapter, "_auto_tts_default"): - adapter._auto_tts_default = auto_tts_default + # A2A never inherits the global speak default. The flag is written by + # Desktop "Read replies aloud" / voice.auto_tts and is meant for human + # chat surfaces; an agent peer's replies must stay text (its adapter has + # no native send_voice, so auto-TTS would synthesize an MP3 and then + # fail delivery) — /voice scoped modes never applied to A2A anyway. + if platform.value == "a2a": + adapter._auto_tts_default = False + else: + adapter._auto_tts_default = auto_tts_default prefix = self._voice_key(platform, "", profile=getattr(adapter, "_owner_profile", None)) for chats, modes in chat_sets: chats.clear() @@ -296,6 +304,13 @@ class GatewayVoiceMixin: — UNLESS streaming consumed the response (already_sent): then the runner must do it.""" if not response or response.startswith("Error:"): return False + # A2A is agent-to-agent text. The adapter has no native send_voice, so + # global voice.auto_tts (Desktop "Read replies aloud") would synthesize + # an MP3 and then fail delivery with "Couldn't deliver the audio + # attachment." — the peer sees the failure instead of the text reply the + # agent already produced (#90103). Keep /voice scoped to human platforms. + if getattr(event.source.platform, "value", None) == "a2a": + return False chat_id = event.source.chat_id voice_mode = self._voice_mode.get(self._voice_key_for_source(event.source)) is_voice_input = event.message_type == MessageType.VOICE diff --git a/tests/gateway/test_auto_voice_reply_format.py b/tests/gateway/test_auto_voice_reply_format.py index bff36f0005..3c9476118b 100644 --- a/tests/gateway/test_auto_voice_reply_format.py +++ b/tests/gateway/test_auto_voice_reply_format.py @@ -93,6 +93,56 @@ class TestAutoVoiceReplyFormat: voice_event = _make_event(Platform.TELEGRAM, chat_id="123", message_type=MessageType.VOICE) assert runner._should_send_voice_reply(voice_event, "hello", [], already_sent=True) is True + def test_should_send_voice_reply_a2a_ignores_global_auto_tts(self): + """A2A text tasks must stay text even when voice.auto_tts is on. + + Desktop Read-replies-aloud writes the global default. The runner used + that default for every adapter with no /voice mode, including A2A, + which cannot deliver native audio: the synthesized MP3 failed delivery + and the peer saw "Couldn't deliver the audio attachment." instead of + the agent's text reply (#90103). + """ + runner = _make_runner() + a2a = Platform("a2a") + adapter = _make_adapter(a2a) + adapter._should_auto_tts_for_chat = MagicMock(return_value=True) + runner.adapters[a2a] = adapter + event = _make_event(a2a, chat_id="ctx-peer") + + assert runner._should_send_voice_reply(event, "audit findings", []) is False + + # The same global default still voices a human platform. + telegram_adapter = _make_adapter(Platform.TELEGRAM) + telegram_adapter._should_auto_tts_for_chat = MagicMock(return_value=True) + runner.adapters[Platform.TELEGRAM] = telegram_adapter + telegram = _make_event(Platform.TELEGRAM, chat_id="999") + + assert runner._should_send_voice_reply(telegram, "hello", []) is True + + def test_sync_voice_mode_state_never_inherits_global_auto_tts_for_a2a(self): + """The adapter-side default must match the runner-side skip (#90103). + + voice.auto_tts is synced onto every adapter at connect; for A2A the + base adapter's own auto-TTS gate would otherwise read a speak default + the platform cannot honor. + """ + runner = _make_runner() + a2a = Platform("a2a") + a2a_adapter = _make_adapter(a2a) + a2a_adapter._auto_tts_disabled_chats = set() + a2a_adapter._auto_tts_enabled_chats = set() + + telegram_adapter = _make_adapter(Platform.TELEGRAM) + telegram_adapter._auto_tts_disabled_chats = set() + telegram_adapter._auto_tts_enabled_chats = set() + + with patch("hermes_cli.config.load_config", return_value={"voice": {"auto_tts": True}}): + runner._sync_voice_mode_state_to_adapter(a2a_adapter) + runner._sync_voice_mode_state_to_adapter(telegram_adapter) + + assert a2a_adapter._auto_tts_default is False + assert telegram_adapter._auto_tts_default is True + def _make_runner() -> GatewayRunner: with patch("gateway.run.GatewayRunner._load_voice_modes", return_value={}): runner = GatewayRunner.__new__(GatewayRunner)