fix(desktop): Bot chats speak with the Bot's own profile TTS voice
Voice playback resolved its (connection, profile) scope from the ACTIVE gateway profile (`getApiRequestProfile()`) on every leg of the ladder — client-direct `fetchVoiceClientConfig`, the speak-stream WS URL and the `/api/audio/speak` relay — so every Bot spoke with the active profile's voice regardless of its own `tts.*` config. `VoicePlaybackOptions.profile` now carries the speaking session's owner profile; `fetchVoiceClientConfig`/`directTtsConfig`, `resolveSpeakStreamUrl` and `speakText` accept it and fall back to the active profile when absent. The session tile publishes `ownerRoute.targetProfile || ownerRoute.profile` on its ComposerScope, and the three speakers (read-aloud button, auto-speak, voice conversation) pass it through. The config cache keys on the owner profile so two Bots never share credentials. Slim redo of #101545 (@FalconOrtiz). Co-authored-by: FalconOrtiz <falcon.ortiz11@gmail.com>
This commit is contained in:
@@ -184,9 +184,11 @@ export function transcribeAudio(dataUrl: string, mimeType?: string): Promise<Aud
|
||||
})
|
||||
}
|
||||
|
||||
export function speakText(text: string): Promise<AudioSpeakResponse> {
|
||||
// `profile` = the speaking session's owner profile (a Bot's own TTS voice);
|
||||
// omitted → the active profile.
|
||||
export function speakText(text: string, profile?: null | string): Promise<AudioSpeakResponse> {
|
||||
return hermesApi<AudioSpeakResponse>({
|
||||
...profileScoped(),
|
||||
...profileScoped(profile || undefined),
|
||||
path: '/api/audio/speak',
|
||||
method: 'POST',
|
||||
body: { text },
|
||||
|
||||
@@ -43,9 +43,9 @@ export function useAutoSpeakReplies({
|
||||
const enabled = useStore($autoSpeakReplies)
|
||||
// Wake on THIS composer's transcript: a tile subscribed to the primary's
|
||||
// would never fire on its own replies (and would fire on someone else's).
|
||||
const { $messages } = useComposerScope()
|
||||
const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply })
|
||||
latest.current = { conversationActive, failureLabel, markSpoken, pendingReply }
|
||||
const { $messages, profile } = useComposerScope()
|
||||
const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply, profile })
|
||||
latest.current = { conversationActive, failureLabel, markSpoken, pendingReply, profile }
|
||||
|
||||
useEffect(() => {
|
||||
if (!enabled) {
|
||||
@@ -57,7 +57,7 @@ export function useAutoSpeakReplies({
|
||||
latest.current.markSpoken()
|
||||
|
||||
const speakLatest = () => {
|
||||
const { conversationActive, failureLabel, markSpoken, pendingReply } = latest.current
|
||||
const { conversationActive, failureLabel, markSpoken, pendingReply, profile } = latest.current
|
||||
|
||||
if (conversationActive || $voicePlayback.get().status !== 'idle') {
|
||||
return
|
||||
@@ -75,7 +75,7 @@ export function useAutoSpeakReplies({
|
||||
// ran in every window, so peers just stay quiet.
|
||||
void ownsAmbientCue(`speak:${reply.id}`).then(owns => {
|
||||
if (owns) {
|
||||
void playSpeechText(reply.text, { messageId: reply.id, source: 'read-aloud' }).catch(error =>
|
||||
void playSpeechText(reply.text, { messageId: reply.id, profile, source: 'read-aloud' }).catch(error =>
|
||||
notifyError(error, failureLabel)
|
||||
)
|
||||
}
|
||||
|
||||
@@ -14,6 +14,8 @@ import { isVoiceStopCommand } from '@/lib/voice-stop-word'
|
||||
import { notify, notifyError } from '@/store/notifications'
|
||||
import { $voicePlayback } from '@/store/voice-playback'
|
||||
|
||||
import { useComposerScope } from '../scope'
|
||||
|
||||
import { useMicRecorder } from './use-mic-recorder'
|
||||
|
||||
export type ConversationStatus = 'idle' | 'listening' | 'transcribing' | 'thinking' | 'speaking'
|
||||
@@ -60,6 +62,11 @@ export function useVoiceConversation({
|
||||
const { t } = useI18n()
|
||||
const voiceCopy = t.notifications.voice
|
||||
const { handle, level } = useMicRecorder(voiceCopy)
|
||||
// The scope's session owner (a Bot's own profile) picks the TTS voice; a
|
||||
// ref keeps the long-lived turn closures below reading the current value.
|
||||
const { profile: ownerProfile } = useComposerScope()
|
||||
const ownerProfileRef = useRef(ownerProfile)
|
||||
ownerProfileRef.current = ownerProfile
|
||||
const [status, setStatus] = useState<ConversationStatus>('idle')
|
||||
const [muted, setMuted] = useState(false)
|
||||
const turnTimeoutRef = useRef<number | null>(null)
|
||||
@@ -459,7 +466,10 @@ export function useVoiceConversation({
|
||||
// this is a safety net for read-aloud-style entries into the loop.
|
||||
ensureBargeMonitor()
|
||||
|
||||
const playback = playSpeechText(response.text, { source: 'voice-conversation' })
|
||||
const playback = playSpeechText(response.text, {
|
||||
profile: ownerProfileRef.current,
|
||||
source: 'voice-conversation'
|
||||
})
|
||||
// playSpeechText performs its normal cleanup synchronously before
|
||||
// returning. Capture the sequence after that internal increment so
|
||||
// only a later, external stop suppresses the next listen cycle.
|
||||
@@ -500,7 +510,7 @@ export function useVoiceConversation({
|
||||
ensureBargeMonitor()
|
||||
|
||||
void (async () => {
|
||||
const session = await startSpeechStream({ source: 'voice-conversation' })
|
||||
const session = await startSpeechStream({ profile: ownerProfileRef.current, source: 'voice-conversation' })
|
||||
|
||||
// The session may resolve after the loop moved on (barge, disable).
|
||||
if (responseIdRef.current !== responseId) {
|
||||
|
||||
@@ -27,6 +27,9 @@ export interface ComposerScope {
|
||||
* keep streaming out of the composer's renders; subscribe only off-render
|
||||
* (auto-speak) where the reply edge is the whole point. */
|
||||
$messages: ReadableAtom<ChatMessage[]>
|
||||
/** Owner profile of this scope's session (a Bot tile runs on the Bot's own
|
||||
* profile). Voice playback synthesizes with it; undefined → active profile. */
|
||||
profile?: null | string
|
||||
/** Focus-bus routing key (`'main'` | `'tile:<id>'`). */
|
||||
target: ComposerTarget
|
||||
}
|
||||
|
||||
@@ -226,9 +226,10 @@ function TileChat({
|
||||
$awaitingInput: sessionAwaitingInput(runtimeId),
|
||||
$messages: view.$messages,
|
||||
attachments,
|
||||
profile: ownerRoute?.targetProfile || ownerRoute?.profile || undefined,
|
||||
target: `tile:${storedSessionId}`
|
||||
}),
|
||||
[attachments, runtimeId, storedSessionId, view.$messages]
|
||||
[attachments, ownerRoute?.profile, ownerRoute?.targetProfile, runtimeId, storedSessionId, view.$messages]
|
||||
)
|
||||
|
||||
// Tile actions must keep the persisted owner route. The ambient gateway hook
|
||||
|
||||
@@ -13,6 +13,7 @@ import { type FC, type ReactNode, useCallback, useContext, useEffect, useMemo, u
|
||||
import { useInRouterContext, useNavigate } from 'react-router'
|
||||
|
||||
import { requestModelMenuToggle } from '@/app/chat/composer/focus'
|
||||
import { useComposerScope } from '@/app/chat/composer/scope'
|
||||
import { useSessionView } from '@/app/chat/session-view'
|
||||
import { SETTINGS_ROUTE } from '@/app/routes'
|
||||
import { dispatchedTo } from '@/components/assistant-ui/thread/agent-delivery'
|
||||
@@ -1043,6 +1044,8 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get
|
||||
const voicePlayback = useStore($voicePlayback)
|
||||
const view = useSessionView()
|
||||
const sessionId = useStore(view.$runtimeId)
|
||||
// A Bot tile's session owns its own profile → its own TTS voice.
|
||||
const { profile } = useComposerScope()
|
||||
|
||||
const readAloudStatus =
|
||||
voicePlayback.source === 'read-aloud' && voicePlayback.messageId === messageId ? voicePlayback.status : 'idle'
|
||||
@@ -1061,12 +1064,12 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get
|
||||
}
|
||||
|
||||
try {
|
||||
await playSpeechText(text, { messageId, source: 'read-aloud' })
|
||||
await playSpeechText(text, { messageId, profile, source: 'read-aloud' })
|
||||
markAssistantIdSpoken(sessionId, view.$messages.get(), messageId)
|
||||
} catch (error) {
|
||||
notifyError(error, copy.readAloudFailed)
|
||||
}
|
||||
}, [copy.readAloudFailed, getText, messageId, sessionId, view.$messages])
|
||||
}, [copy.readAloudFailed, getText, messageId, profile, sessionId, view.$messages])
|
||||
|
||||
return (
|
||||
<TooltipIconButton
|
||||
|
||||
@@ -77,6 +77,21 @@ describe('fetchVoiceClientConfig', () => {
|
||||
expect(api).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
|
||||
// A Bot chat runs on the Bot's own profile: two Bots open beside the active
|
||||
// profile must each fetch THEIR profile's TTS config, and a plain chat keeps
|
||||
// the active-profile fallback (#100864).
|
||||
it('resolves per-Bot TTS config by the owner profile, keeping the active-profile fallback', async () => {
|
||||
const api = mockDesktopApi({ ok: true, stt: directStt, tts: relay })
|
||||
setApiRequestProfile('research')
|
||||
|
||||
await fetchVoiceClientConfig('bot-rachel')
|
||||
await fetchVoiceClientConfig('bot-adam')
|
||||
await fetchVoiceClientConfig()
|
||||
|
||||
const profiles = api.mock.calls.map(([request]) => (request as { profile?: string }).profile)
|
||||
expect(profiles).toEqual(['bot-rachel', 'bot-adam', 'research'])
|
||||
})
|
||||
|
||||
it('resolves null on an older backend without the endpoint', async () => {
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
|
||||
@@ -70,8 +70,10 @@ const STT_REQUEST_TIMEOUT_MS = 60_000
|
||||
let cached: { key: string; at: number; config: VoiceClientConfig } | null = null
|
||||
let inflight: { key: string; promise: Promise<null | VoiceClientConfig> } | null = null
|
||||
|
||||
function scopeKey(): string {
|
||||
return `${getApiRequestConnection() ?? 'local'}::${getApiRequestProfile() ?? 'default'}`
|
||||
// `profile` is the session OWNER's profile (a Bot chat runs on its own
|
||||
// profile with its own TTS voice); undefined/null → the active profile.
|
||||
function scopeKey(profile?: null | string): string {
|
||||
return `${getApiRequestConnection() ?? 'local'}::${profile || getApiRequestProfile() || 'default'}`
|
||||
}
|
||||
|
||||
/** Drop cached credentials (used by tests; scope changes rotate the key). */
|
||||
@@ -80,8 +82,8 @@ export function clearVoiceClientConfigCache(): void {
|
||||
inflight = null
|
||||
}
|
||||
|
||||
export async function fetchVoiceClientConfig(): Promise<null | VoiceClientConfig> {
|
||||
const key = scopeKey()
|
||||
export async function fetchVoiceClientConfig(profile?: null | string): Promise<null | VoiceClientConfig> {
|
||||
const key = scopeKey(profile)
|
||||
|
||||
if (cached && cached.key === key && Date.now() - cached.at < CONFIG_TTL_MS) {
|
||||
return cached.config
|
||||
@@ -97,7 +99,7 @@ export async function fetchVoiceClientConfig(): Promise<null | VoiceClientConfig
|
||||
// profile — the same routing every relay audio call uses, so the
|
||||
// config comes from the backend the user is actually talking to.
|
||||
const response = await hermesApi<{ ok: boolean } & VoiceClientConfig>({
|
||||
...profileScoped(),
|
||||
...profileScoped(profile || undefined),
|
||||
path: '/api/audio/voice-config'
|
||||
})
|
||||
|
||||
@@ -316,8 +318,8 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise<null | s
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Resolve the profile's TTS config when it is client-callable, else null. */
|
||||
export async function directTtsConfig(): Promise<DirectTtsConfig | null> {
|
||||
const config = await fetchVoiceClientConfig()
|
||||
export async function directTtsConfig(profile?: null | string): Promise<DirectTtsConfig | null> {
|
||||
const config = await fetchVoiceClientConfig(profile)
|
||||
|
||||
return config?.tts && config.tts.mode === 'direct' ? config.tts : null
|
||||
}
|
||||
|
||||
@@ -71,6 +71,15 @@ describe('resolveSpeakStreamUrl', () => {
|
||||
expect(getConnectionFor).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it("dials the speaking session's owner profile ahead of the active profile", async () => {
|
||||
setApiRequestProfile('research')
|
||||
|
||||
const url = await resolveSpeakStreamUrl('bot-adam')
|
||||
|
||||
expect(url).toContain('profile=bot-adam')
|
||||
expect(getConnection).toHaveBeenCalledWith('bot-adam')
|
||||
})
|
||||
|
||||
it('preserves a backend-namespace profile already minted into the ws URL', async () => {
|
||||
// SSH remoteProfile aliasing / sharedRemote scoping: the registry mint
|
||||
// writes the BACKEND's profile name into the URL. The desktop-side
|
||||
|
||||
@@ -73,6 +73,9 @@ function currentState(
|
||||
|
||||
export interface VoicePlaybackOptions {
|
||||
messageId?: string | null
|
||||
/** Owner profile of the speaking session (a Bot chat synthesizes with its
|
||||
* own profile's TTS voice). Omitted → the active profile. */
|
||||
profile?: null | string
|
||||
source: VoicePlaybackSource
|
||||
}
|
||||
|
||||
@@ -105,7 +108,7 @@ export function stopVoicePlayback() {
|
||||
|
||||
/** Exported for tests: the (connection, profile) routing contract below is
|
||||
* exactly what broke in the desktop-remote voice report — keep it pinned. */
|
||||
export async function resolveSpeakStreamUrl(): Promise<null | string> {
|
||||
export async function resolveSpeakStreamUrl(ownerProfile?: null | string): Promise<null | string> {
|
||||
const desktop = window.hermesDesktop
|
||||
|
||||
if (!desktop?.getConnection) {
|
||||
@@ -123,7 +126,7 @@ export async function resolveSpeakStreamUrl(): Promise<null | string> {
|
||||
// replies would synthesize with the local (often unconfigured) TTS
|
||||
// instead of the profile the user is actually talking to (#90051-adjacent
|
||||
// desktop-remote voice report, Aug 2026).
|
||||
const profile = getApiRequestProfile()
|
||||
const profile = ownerProfile || getApiRequestProfile()
|
||||
const connectionId = getApiRequestConnection()
|
||||
|
||||
// Both awaits below are IPC round-trips into the main process with no
|
||||
@@ -522,7 +525,7 @@ function openSpeechStream(wsUrl: string, options: VoicePlaybackOptions): SpeechS
|
||||
* `playSpeechText`).
|
||||
*/
|
||||
export async function startSpeechStream(options: VoicePlaybackOptions): Promise<null | SpeechStreamSession> {
|
||||
const direct = await directTtsConfig().catch(() => null)
|
||||
const direct = await directTtsConfig(options.profile).catch(() => null)
|
||||
|
||||
if (direct) {
|
||||
stopVoicePlayback()
|
||||
@@ -539,7 +542,7 @@ export async function startSpeechStream(options: VoicePlaybackOptions): Promise<
|
||||
return session
|
||||
}
|
||||
|
||||
const wsUrl = await resolveSpeakStreamUrl()
|
||||
const wsUrl = await resolveSpeakStreamUrl(options.profile)
|
||||
|
||||
if (!wsUrl) {
|
||||
return null
|
||||
@@ -573,7 +576,7 @@ async function playSpeechDataUrl(
|
||||
options: VoicePlaybackOptions,
|
||||
isCurrent: () => boolean
|
||||
): Promise<boolean> {
|
||||
const response = await speakText(speakableText)
|
||||
const response = await speakText(speakableText, options.profile)
|
||||
|
||||
if (!isCurrent()) {
|
||||
return false
|
||||
@@ -669,7 +672,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions
|
||||
try {
|
||||
// Ladder: client-direct synthesis (profile's own TTS, no gateway audio
|
||||
// hop) → streaming WS relay → POST data-URL fallback.
|
||||
const direct = await directTtsConfig().catch(() => null)
|
||||
const direct = await directTtsConfig(options.profile).catch(() => null)
|
||||
|
||||
if (direct && isCurrent()) {
|
||||
const session = openClientDirectSpeechSession(direct, options)
|
||||
@@ -693,7 +696,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions
|
||||
return false
|
||||
}
|
||||
|
||||
const streamUrl = await resolveSpeakStreamUrl()
|
||||
const streamUrl = await resolveSpeakStreamUrl(options.profile)
|
||||
|
||||
if (streamUrl && isCurrent()) {
|
||||
const outcome = await playSpeechStream(streamUrl, speakableText, options)
|
||||
|
||||
Reference in New Issue
Block a user