From f3d98d4fda3d88f8d1fb88e80fc82a8ba80e9632 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sun, 20 Sep 2026 00:00:24 -0700 Subject: [PATCH] fix(desktop): Bot chats speak with the Bot's own profile TTS voice MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Voice playback resolved its (connection, profile) scope from the ACTIVE gateway profile (`getApiRequestProfile()`) on every leg of the ladder — client-direct `fetchVoiceClientConfig`, the speak-stream WS URL and the `/api/audio/speak` relay — so every Bot spoke with the active profile's voice regardless of its own `tts.*` config. `VoicePlaybackOptions.profile` now carries the speaking session's owner profile; `fetchVoiceClientConfig`/`directTtsConfig`, `resolveSpeakStreamUrl` and `speakText` accept it and fall back to the active profile when absent. The session tile publishes `ownerRoute.targetProfile || ownerRoute.profile` on its ComposerScope, and the three speakers (read-aloud button, auto-speak, voice conversation) pass it through. The config cache keys on the owner profile so two Bots never share credentials. Slim redo of #101545 (@FalconOrtiz). Co-authored-by: FalconOrtiz --- apps/desktop/src/api/system.ts | 6 ++++-- .../composer/hooks/use-auto-speak-replies.ts | 10 +++++----- .../composer/hooks/use-voice-conversation.ts | 14 ++++++++++++-- apps/desktop/src/app/chat/composer/scope.tsx | 3 +++ apps/desktop/src/app/chat/session-tile.tsx | 3 ++- .../assistant-ui/thread/assistant-message.tsx | 7 +++++-- .../desktop/src/lib/voice-client-direct.test.ts | 15 +++++++++++++++ apps/desktop/src/lib/voice-client-direct.ts | 16 +++++++++------- .../src/lib/voice-playback.routing.test.ts | 9 +++++++++ apps/desktop/src/lib/voice-playback.ts | 17 ++++++++++------- 10 files changed, 74 insertions(+), 26 deletions(-) diff --git a/apps/desktop/src/api/system.ts b/apps/desktop/src/api/system.ts index ba2e0bbcca..4d70299675 100644 --- a/apps/desktop/src/api/system.ts +++ b/apps/desktop/src/api/system.ts @@ -184,9 +184,11 @@ export function transcribeAudio(dataUrl: string, mimeType?: string): Promise { +// `profile` = the speaking session's owner profile (a Bot's own TTS voice); +// omitted → the active profile. +export function speakText(text: string, profile?: null | string): Promise { return hermesApi({ - ...profileScoped(), + ...profileScoped(profile || undefined), path: '/api/audio/speak', method: 'POST', body: { text }, diff --git a/apps/desktop/src/app/chat/composer/hooks/use-auto-speak-replies.ts b/apps/desktop/src/app/chat/composer/hooks/use-auto-speak-replies.ts index aa657ca829..cee46bae5f 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-auto-speak-replies.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-auto-speak-replies.ts @@ -43,9 +43,9 @@ export function useAutoSpeakReplies({ const enabled = useStore($autoSpeakReplies) // Wake on THIS composer's transcript: a tile subscribed to the primary's // would never fire on its own replies (and would fire on someone else's). - const { $messages } = useComposerScope() - const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply }) - latest.current = { conversationActive, failureLabel, markSpoken, pendingReply } + const { $messages, profile } = useComposerScope() + const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply, profile }) + latest.current = { conversationActive, failureLabel, markSpoken, pendingReply, profile } useEffect(() => { if (!enabled) { @@ -57,7 +57,7 @@ export function useAutoSpeakReplies({ latest.current.markSpoken() const speakLatest = () => { - const { conversationActive, failureLabel, markSpoken, pendingReply } = latest.current + const { conversationActive, failureLabel, markSpoken, pendingReply, profile } = latest.current if (conversationActive || $voicePlayback.get().status !== 'idle') { return @@ -75,7 +75,7 @@ export function useAutoSpeakReplies({ // ran in every window, so peers just stay quiet. void ownsAmbientCue(`speak:${reply.id}`).then(owns => { if (owns) { - void playSpeechText(reply.text, { messageId: reply.id, source: 'read-aloud' }).catch(error => + void playSpeechText(reply.text, { messageId: reply.id, profile, source: 'read-aloud' }).catch(error => notifyError(error, failureLabel) ) } diff --git a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts index 55978c4284..2f0df3abdb 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts @@ -14,6 +14,8 @@ import { isVoiceStopCommand } from '@/lib/voice-stop-word' import { notify, notifyError } from '@/store/notifications' import { $voicePlayback } from '@/store/voice-playback' +import { useComposerScope } from '../scope' + import { useMicRecorder } from './use-mic-recorder' export type ConversationStatus = 'idle' | 'listening' | 'transcribing' | 'thinking' | 'speaking' @@ -60,6 +62,11 @@ export function useVoiceConversation({ const { t } = useI18n() const voiceCopy = t.notifications.voice const { handle, level } = useMicRecorder(voiceCopy) + // The scope's session owner (a Bot's own profile) picks the TTS voice; a + // ref keeps the long-lived turn closures below reading the current value. + const { profile: ownerProfile } = useComposerScope() + const ownerProfileRef = useRef(ownerProfile) + ownerProfileRef.current = ownerProfile const [status, setStatus] = useState('idle') const [muted, setMuted] = useState(false) const turnTimeoutRef = useRef(null) @@ -459,7 +466,10 @@ export function useVoiceConversation({ // this is a safety net for read-aloud-style entries into the loop. ensureBargeMonitor() - const playback = playSpeechText(response.text, { source: 'voice-conversation' }) + const playback = playSpeechText(response.text, { + profile: ownerProfileRef.current, + source: 'voice-conversation' + }) // playSpeechText performs its normal cleanup synchronously before // returning. Capture the sequence after that internal increment so // only a later, external stop suppresses the next listen cycle. @@ -500,7 +510,7 @@ export function useVoiceConversation({ ensureBargeMonitor() void (async () => { - const session = await startSpeechStream({ source: 'voice-conversation' }) + const session = await startSpeechStream({ profile: ownerProfileRef.current, source: 'voice-conversation' }) // The session may resolve after the loop moved on (barge, disable). if (responseIdRef.current !== responseId) { diff --git a/apps/desktop/src/app/chat/composer/scope.tsx b/apps/desktop/src/app/chat/composer/scope.tsx index a14b53ca42..ea77eacac5 100644 --- a/apps/desktop/src/app/chat/composer/scope.tsx +++ b/apps/desktop/src/app/chat/composer/scope.tsx @@ -27,6 +27,9 @@ export interface ComposerScope { * keep streaming out of the composer's renders; subscribe only off-render * (auto-speak) where the reply edge is the whole point. */ $messages: ReadableAtom + /** Owner profile of this scope's session (a Bot tile runs on the Bot's own + * profile). Voice playback synthesizes with it; undefined → active profile. */ + profile?: null | string /** Focus-bus routing key (`'main'` | `'tile:'`). */ target: ComposerTarget } diff --git a/apps/desktop/src/app/chat/session-tile.tsx b/apps/desktop/src/app/chat/session-tile.tsx index 0ee974ed59..d05829774b 100644 --- a/apps/desktop/src/app/chat/session-tile.tsx +++ b/apps/desktop/src/app/chat/session-tile.tsx @@ -226,9 +226,10 @@ function TileChat({ $awaitingInput: sessionAwaitingInput(runtimeId), $messages: view.$messages, attachments, + profile: ownerRoute?.targetProfile || ownerRoute?.profile || undefined, target: `tile:${storedSessionId}` }), - [attachments, runtimeId, storedSessionId, view.$messages] + [attachments, ownerRoute?.profile, ownerRoute?.targetProfile, runtimeId, storedSessionId, view.$messages] ) // Tile actions must keep the persisted owner route. The ambient gateway hook diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index 75773f88e9..d7d6ab54be 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -13,6 +13,7 @@ import { type FC, type ReactNode, useCallback, useContext, useEffect, useMemo, u import { useInRouterContext, useNavigate } from 'react-router' import { requestModelMenuToggle } from '@/app/chat/composer/focus' +import { useComposerScope } from '@/app/chat/composer/scope' import { useSessionView } from '@/app/chat/session-view' import { SETTINGS_ROUTE } from '@/app/routes' import { dispatchedTo } from '@/components/assistant-ui/thread/agent-delivery' @@ -1043,6 +1044,8 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get const voicePlayback = useStore($voicePlayback) const view = useSessionView() const sessionId = useStore(view.$runtimeId) + // A Bot tile's session owns its own profile → its own TTS voice. + const { profile } = useComposerScope() const readAloudStatus = voicePlayback.source === 'read-aloud' && voicePlayback.messageId === messageId ? voicePlayback.status : 'idle' @@ -1061,12 +1064,12 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get } try { - await playSpeechText(text, { messageId, source: 'read-aloud' }) + await playSpeechText(text, { messageId, profile, source: 'read-aloud' }) markAssistantIdSpoken(sessionId, view.$messages.get(), messageId) } catch (error) { notifyError(error, copy.readAloudFailed) } - }, [copy.readAloudFailed, getText, messageId, sessionId, view.$messages]) + }, [copy.readAloudFailed, getText, messageId, profile, sessionId, view.$messages]) return ( { expect(api).toHaveBeenCalledTimes(2) }) + // A Bot chat runs on the Bot's own profile: two Bots open beside the active + // profile must each fetch THEIR profile's TTS config, and a plain chat keeps + // the active-profile fallback (#100864). + it('resolves per-Bot TTS config by the owner profile, keeping the active-profile fallback', async () => { + const api = mockDesktopApi({ ok: true, stt: directStt, tts: relay }) + setApiRequestProfile('research') + + await fetchVoiceClientConfig('bot-rachel') + await fetchVoiceClientConfig('bot-adam') + await fetchVoiceClientConfig() + + const profiles = api.mock.calls.map(([request]) => (request as { profile?: string }).profile) + expect(profiles).toEqual(['bot-rachel', 'bot-adam', 'research']) + }) + it('resolves null on an older backend without the endpoint', async () => { Object.defineProperty(window, 'hermesDesktop', { configurable: true, diff --git a/apps/desktop/src/lib/voice-client-direct.ts b/apps/desktop/src/lib/voice-client-direct.ts index 972c8da742..2c5b80636f 100644 --- a/apps/desktop/src/lib/voice-client-direct.ts +++ b/apps/desktop/src/lib/voice-client-direct.ts @@ -70,8 +70,10 @@ const STT_REQUEST_TIMEOUT_MS = 60_000 let cached: { key: string; at: number; config: VoiceClientConfig } | null = null let inflight: { key: string; promise: Promise } | null = null -function scopeKey(): string { - return `${getApiRequestConnection() ?? 'local'}::${getApiRequestProfile() ?? 'default'}` +// `profile` is the session OWNER's profile (a Bot chat runs on its own +// profile with its own TTS voice); undefined/null → the active profile. +function scopeKey(profile?: null | string): string { + return `${getApiRequestConnection() ?? 'local'}::${profile || getApiRequestProfile() || 'default'}` } /** Drop cached credentials (used by tests; scope changes rotate the key). */ @@ -80,8 +82,8 @@ export function clearVoiceClientConfigCache(): void { inflight = null } -export async function fetchVoiceClientConfig(): Promise { - const key = scopeKey() +export async function fetchVoiceClientConfig(profile?: null | string): Promise { + const key = scopeKey(profile) if (cached && cached.key === key && Date.now() - cached.at < CONFIG_TTL_MS) { return cached.config @@ -97,7 +99,7 @@ export async function fetchVoiceClientConfig(): Promise({ - ...profileScoped(), + ...profileScoped(profile || undefined), path: '/api/audio/voice-config' }) @@ -316,8 +318,8 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise { - const config = await fetchVoiceClientConfig() +export async function directTtsConfig(profile?: null | string): Promise { + const config = await fetchVoiceClientConfig(profile) return config?.tts && config.tts.mode === 'direct' ? config.tts : null } diff --git a/apps/desktop/src/lib/voice-playback.routing.test.ts b/apps/desktop/src/lib/voice-playback.routing.test.ts index 90e70ad4f9..8e7244cfd3 100644 --- a/apps/desktop/src/lib/voice-playback.routing.test.ts +++ b/apps/desktop/src/lib/voice-playback.routing.test.ts @@ -71,6 +71,15 @@ describe('resolveSpeakStreamUrl', () => { expect(getConnectionFor).not.toHaveBeenCalled() }) + it("dials the speaking session's owner profile ahead of the active profile", async () => { + setApiRequestProfile('research') + + const url = await resolveSpeakStreamUrl('bot-adam') + + expect(url).toContain('profile=bot-adam') + expect(getConnection).toHaveBeenCalledWith('bot-adam') + }) + it('preserves a backend-namespace profile already minted into the ws URL', async () => { // SSH remoteProfile aliasing / sharedRemote scoping: the registry mint // writes the BACKEND's profile name into the URL. The desktop-side diff --git a/apps/desktop/src/lib/voice-playback.ts b/apps/desktop/src/lib/voice-playback.ts index 8fc85065d9..8b5c47e10b 100644 --- a/apps/desktop/src/lib/voice-playback.ts +++ b/apps/desktop/src/lib/voice-playback.ts @@ -73,6 +73,9 @@ function currentState( export interface VoicePlaybackOptions { messageId?: string | null + /** Owner profile of the speaking session (a Bot chat synthesizes with its + * own profile's TTS voice). Omitted → the active profile. */ + profile?: null | string source: VoicePlaybackSource } @@ -105,7 +108,7 @@ export function stopVoicePlayback() { /** Exported for tests: the (connection, profile) routing contract below is * exactly what broke in the desktop-remote voice report — keep it pinned. */ -export async function resolveSpeakStreamUrl(): Promise { +export async function resolveSpeakStreamUrl(ownerProfile?: null | string): Promise { const desktop = window.hermesDesktop if (!desktop?.getConnection) { @@ -123,7 +126,7 @@ export async function resolveSpeakStreamUrl(): Promise { // replies would synthesize with the local (often unconfigured) TTS // instead of the profile the user is actually talking to (#90051-adjacent // desktop-remote voice report, Aug 2026). - const profile = getApiRequestProfile() + const profile = ownerProfile || getApiRequestProfile() const connectionId = getApiRequestConnection() // Both awaits below are IPC round-trips into the main process with no @@ -522,7 +525,7 @@ function openSpeechStream(wsUrl: string, options: VoicePlaybackOptions): SpeechS * `playSpeechText`). */ export async function startSpeechStream(options: VoicePlaybackOptions): Promise { - const direct = await directTtsConfig().catch(() => null) + const direct = await directTtsConfig(options.profile).catch(() => null) if (direct) { stopVoicePlayback() @@ -539,7 +542,7 @@ export async function startSpeechStream(options: VoicePlaybackOptions): Promise< return session } - const wsUrl = await resolveSpeakStreamUrl() + const wsUrl = await resolveSpeakStreamUrl(options.profile) if (!wsUrl) { return null @@ -573,7 +576,7 @@ async function playSpeechDataUrl( options: VoicePlaybackOptions, isCurrent: () => boolean ): Promise { - const response = await speakText(speakableText) + const response = await speakText(speakableText, options.profile) if (!isCurrent()) { return false @@ -669,7 +672,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions try { // Ladder: client-direct synthesis (profile's own TTS, no gateway audio // hop) → streaming WS relay → POST data-URL fallback. - const direct = await directTtsConfig().catch(() => null) + const direct = await directTtsConfig(options.profile).catch(() => null) if (direct && isCurrent()) { const session = openClientDirectSpeechSession(direct, options) @@ -693,7 +696,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions return false } - const streamUrl = await resolveSpeakStreamUrl() + const streamUrl = await resolveSpeakStreamUrl(options.profile) if (streamUrl && isCurrent()) { const outcome = await playSpeechStream(streamUrl, speakableText, options)