Files
hermes-agent/tests/gateway/test_auto_voice_reply_format.py
kshitijk4poor ab2f4602de refactor: MessageEvent to gateway/platforms/event.py; ElicitationHandler takes a call_context thunk
Breaks the two import cycles that forced Protocol stand-ins in the F821 sweep, so the two
sites now name the real types.

gateway/platforms/event.py (new leaf): MessageType, ProcessingOutcome, MessageEvent moved
out of base.py verbatim. Their only dependency is gateway.session.SessionSource; base.py
imported helpers.py at module level, so helpers could not name MessageEvent. Now
TextBatchAggregator is typed by the real MessageEvent. 249 importers repointed
(`from gateway.platforms.base import` -> `.event`, preserving each import's layout);
gateway.platforms.__init__ re-exports from .event. The three revert-scheduled PLUGIN-COMPAT
pointers that named these symbols (gateway.slash_commands → MessageType, dingtalk → MessageType,
photon → ProcessingOutcome) and their COMPAT_MANIFEST rows now target gateway.platforms.event.
Docs updated: ADDING_A_PLATFORM.md, adding-platform-adapters.md (en + zh-Hans).

tools/mcp_tool_sampling.py: ElicitationHandler no longer holds a back-reference to its
MCPServerTask (mcp_tool imports sampling, so the task type cannot be named there). It only
ever read owner._pending_call_context, so it takes `call_context: Callable[[], Context | None]`
and MCPServerTask passes `lambda: self._pending_call_context`. The consent call is one
`functools.partial`, run directly or inside the captured Context.

ty on the 11 touched production files vs origin/main: 0 new diagnostics, 14 resolved.
(The one `source: SessionSource = None` diagnostic moves with the class; typing it Optional
exposes ~60 unguarded call sites — separate follow-up.)

Tests: tests/gateway + tests/plugins + tests/tools + touched files, 18,235 passed; the 31
failures reproduce identically on origin/main (macOS /private/tmp, systemd socket,
long-path fixtures, live-service tests).
2026-09-07 22:47:33 +05:30

123 lines
4.6 KiB
Python

"""Tests for gateway auto-TTS voice reply audio format selection."""
import json
from pathlib import Path
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from gateway.config import Platform
from gateway.platforms.event import MessageEvent, MessageType
from gateway.run import GatewayRunner
from gateway.session import SessionSource
class TestAutoVoiceReplyFormat:
@pytest.mark.asyncio
@pytest.mark.parametrize(
"platform",
[Platform.MATRIX, Platform.FEISHU, Platform.WHATSAPP, Platform.SIGNAL],
)
async def test_opus_platform_auto_voice_reply_requests_ogg(self, platform):
"""Every OPUS_VOICE_PLATFORMS member gets an explicit .ogg output path.
Regression for #14841 (Matrix) / #45557 (Feishu): _send_voice_reply
hardcoded .ogg for Telegram only, so Matrix/Feishu voice replies were
synthesized as MP3 and delivered as plain attachments instead of
native voice bubbles.
"""
runner = _make_runner()
adapter = _make_adapter(platform)
runner.adapters[platform] = adapter
event = _make_event(platform)
requested_paths = []
def fake_tts(*, text, output_path):
requested_paths.append(output_path)
Path(output_path).parent.mkdir(parents=True, exist_ok=True)
Path(output_path).write_bytes(b"fake ogg opus")
return json.dumps({
"success": True,
"file_path": output_path,
"provider": "gemini",
"voice_compatible": True,
})
with patch("tools.tts_tool.text_to_speech_tool", side_effect=fake_tts):
await runner._send_voice_reply(event, "hello from auto tts")
assert requested_paths and requested_paths[0].endswith(".ogg")
adapter.send_voice.assert_awaited_once()
assert adapter.send_voice.await_args.kwargs["audio_path"].endswith(".ogg")
def test_should_send_voice_reply_streamed_global_auto_tts_fires(self):
"""Streamed reply + global voice.auto_tts (no /voice opt-in) sends voice.
Regression for the #51867/#23983 remainder: when streaming consumed
the text, the base adapter's auto-TTS gets text_content=None, and the
runner path used to consult only self._voice_mode — so a chat relying
purely on the global voice.auto_tts default silently lost its voice
reply.
"""
runner = _make_runner()
adapter = _make_adapter(Platform.TELEGRAM)
adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
runner.adapters[Platform.TELEGRAM] = adapter
voice_event = _make_event(
Platform.TELEGRAM, chat_id="123", message_type=MessageType.VOICE
)
assert runner._should_send_voice_reply(
voice_event, "hello", [], already_sent=True
) is True
def test_should_send_voice_reply_voice_only_still_requires_voice_input(self):
"""Explicit voice_only must not widen to text input (#73508 regression).
Persisted voice_only mode is synced into the adapter as an explicit
auto-TTS opt-in, so adapter_auto_tts is True for this chat. The
chat-level mode stays authoritative: text input gets no voice reply,
voice input still does.
"""
runner = _make_runner()
runner._voice_mode["telegram:123"] = "voice_only"
adapter = _make_adapter(Platform.TELEGRAM)
adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
runner.adapters[Platform.TELEGRAM] = adapter
event = _make_event(Platform.TELEGRAM, chat_id="123")
assert runner._should_send_voice_reply(event, "hello", []) is False
voice_event = _make_event(Platform.TELEGRAM, chat_id="123", message_type=MessageType.VOICE)
assert runner._should_send_voice_reply(voice_event, "hello", [], already_sent=True) is True
def _make_runner() -> GatewayRunner:
with patch("gateway.run.GatewayRunner._load_voice_modes", return_value={}):
runner = GatewayRunner.__new__(GatewayRunner)
runner._voice_mode = {}
runner.adapters = {}
return runner
def _make_adapter(platform: Platform) -> MagicMock:
adapter = MagicMock()
adapter.platform = platform
adapter.send_voice = AsyncMock()
return adapter
def _make_event(platform: Platform, chat_id: str = "123", message_type: MessageType = MessageType.TEXT) -> MessageEvent:
return MessageEvent(
text="trigger",
source=SessionSource(
platform=platform,
chat_id=chat_id,
user_id="u1",
user_name="User",
),
message_type=message_type,
message_id="456",
)