fix(gateway): do not auto-TTS A2A replies
voice.auto_tts (flipped globally by Desktop "Read replies aloud") made the runner synthesize a spoken reply for every A2A text task. The A2A adapter has no native send_voice, so delivery fell back to the media notice — the peer received "Couldn't deliver the audio attachment." instead of the text the agent had already produced (#90103). Inbound A2A is MessageType.TEXT; the base adapter's own auto-TTS gate keys on MessageType.VOICE, but the runner's fallback branch (voice_mode is None and adapter_auto_tts) had no platform gate at all. Two aligned gates, both field-tested by the reporter: - _should_send_voice_reply returns False for platform 'a2a' before any mode/config consultation — the text reply lands normally. - _sync_voice_mode_state_to_adapter never pushes the global speak default onto the A2A adapter, so the adapter-side path cannot regress it either. /voice on|tts|off scoped behavior is untouched (persisted per platform:chat_id, still honored for human platforms). Tests pin the pair the issue asked for: A2A + global auto-TTS skips, Telegram with the same default still voices, and the sync-side skip. Re-implemented from PR #90121 on the GatewayVoiceMixin split (author credited). The Desktop-scoped read-aloud preference is a separate design change, not folded in. Fixes #90103
This commit is contained in:
@@ -122,7 +122,15 @@ class GatewayVoiceMixin:
|
||||
except Exception:
|
||||
auto_tts_default = False
|
||||
if hasattr(adapter, "_auto_tts_default"):
|
||||
adapter._auto_tts_default = auto_tts_default
|
||||
# A2A never inherits the global speak default. The flag is written by
|
||||
# Desktop "Read replies aloud" / voice.auto_tts and is meant for human
|
||||
# chat surfaces; an agent peer's replies must stay text (its adapter has
|
||||
# no native send_voice, so auto-TTS would synthesize an MP3 and then
|
||||
# fail delivery) — /voice scoped modes never applied to A2A anyway.
|
||||
if platform.value == "a2a":
|
||||
adapter._auto_tts_default = False
|
||||
else:
|
||||
adapter._auto_tts_default = auto_tts_default
|
||||
prefix = self._voice_key(platform, "", profile=getattr(adapter, "_owner_profile", None))
|
||||
for chats, modes in chat_sets:
|
||||
chats.clear()
|
||||
@@ -296,6 +304,13 @@ class GatewayVoiceMixin:
|
||||
— UNLESS streaming consumed the response (already_sent): then the runner must do it."""
|
||||
if not response or response.startswith("Error:"):
|
||||
return False
|
||||
# A2A is agent-to-agent text. The adapter has no native send_voice, so
|
||||
# global voice.auto_tts (Desktop "Read replies aloud") would synthesize
|
||||
# an MP3 and then fail delivery with "Couldn't deliver the audio
|
||||
# attachment." — the peer sees the failure instead of the text reply the
|
||||
# agent already produced (#90103). Keep /voice scoped to human platforms.
|
||||
if getattr(event.source.platform, "value", None) == "a2a":
|
||||
return False
|
||||
chat_id = event.source.chat_id
|
||||
voice_mode = self._voice_mode.get(self._voice_key_for_source(event.source))
|
||||
is_voice_input = event.message_type == MessageType.VOICE
|
||||
|
||||
@@ -93,6 +93,56 @@ class TestAutoVoiceReplyFormat:
|
||||
voice_event = _make_event(Platform.TELEGRAM, chat_id="123", message_type=MessageType.VOICE)
|
||||
assert runner._should_send_voice_reply(voice_event, "hello", [], already_sent=True) is True
|
||||
|
||||
def test_should_send_voice_reply_a2a_ignores_global_auto_tts(self):
|
||||
"""A2A text tasks must stay text even when voice.auto_tts is on.
|
||||
|
||||
Desktop Read-replies-aloud writes the global default. The runner used
|
||||
that default for every adapter with no /voice mode, including A2A,
|
||||
which cannot deliver native audio: the synthesized MP3 failed delivery
|
||||
and the peer saw "Couldn't deliver the audio attachment." instead of
|
||||
the agent's text reply (#90103).
|
||||
"""
|
||||
runner = _make_runner()
|
||||
a2a = Platform("a2a")
|
||||
adapter = _make_adapter(a2a)
|
||||
adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
|
||||
runner.adapters[a2a] = adapter
|
||||
event = _make_event(a2a, chat_id="ctx-peer")
|
||||
|
||||
assert runner._should_send_voice_reply(event, "audit findings", []) is False
|
||||
|
||||
# The same global default still voices a human platform.
|
||||
telegram_adapter = _make_adapter(Platform.TELEGRAM)
|
||||
telegram_adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
|
||||
runner.adapters[Platform.TELEGRAM] = telegram_adapter
|
||||
telegram = _make_event(Platform.TELEGRAM, chat_id="999")
|
||||
|
||||
assert runner._should_send_voice_reply(telegram, "hello", []) is True
|
||||
|
||||
def test_sync_voice_mode_state_never_inherits_global_auto_tts_for_a2a(self):
|
||||
"""The adapter-side default must match the runner-side skip (#90103).
|
||||
|
||||
voice.auto_tts is synced onto every adapter at connect; for A2A the
|
||||
base adapter's own auto-TTS gate would otherwise read a speak default
|
||||
the platform cannot honor.
|
||||
"""
|
||||
runner = _make_runner()
|
||||
a2a = Platform("a2a")
|
||||
a2a_adapter = _make_adapter(a2a)
|
||||
a2a_adapter._auto_tts_disabled_chats = set()
|
||||
a2a_adapter._auto_tts_enabled_chats = set()
|
||||
|
||||
telegram_adapter = _make_adapter(Platform.TELEGRAM)
|
||||
telegram_adapter._auto_tts_disabled_chats = set()
|
||||
telegram_adapter._auto_tts_enabled_chats = set()
|
||||
|
||||
with patch("hermes_cli.config.load_config", return_value={"voice": {"auto_tts": True}}):
|
||||
runner._sync_voice_mode_state_to_adapter(a2a_adapter)
|
||||
runner._sync_voice_mode_state_to_adapter(telegram_adapter)
|
||||
|
||||
assert a2a_adapter._auto_tts_default is False
|
||||
assert telegram_adapter._auto_tts_default is True
|
||||
|
||||
def _make_runner() -> GatewayRunner:
|
||||
with patch("gateway.run.GatewayRunner._load_voice_modes", return_value={}):
|
||||
runner = GatewayRunner.__new__(GatewayRunner)
|
||||
|
||||
Reference in New Issue
Block a user