fix(desktop): Bot chats speak with the Bot's own profile TTS voice

Voice playback resolved its (connection, profile) scope from the ACTIVE
gateway profile (`getApiRequestProfile()`) on every leg of the ladder —
client-direct `fetchVoiceClientConfig`, the speak-stream WS URL and the
`/api/audio/speak` relay — so every Bot spoke with the active profile's
voice regardless of its own `tts.*` config.

`VoicePlaybackOptions.profile` now carries the speaking session's owner
profile; `fetchVoiceClientConfig`/`directTtsConfig`, `resolveSpeakStreamUrl`
and `speakText` accept it and fall back to the active profile when absent.
The session tile publishes `ownerRoute.targetProfile || ownerRoute.profile`
on its ComposerScope, and the three speakers (read-aloud button, auto-speak,
voice conversation) pass it through. The config cache keys on the owner
profile so two Bots never share credentials.

Slim redo of #101545 (@FalconOrtiz).

Co-authored-by: FalconOrtiz <falcon.ortiz11@gmail.com>
This commit is contained in:
teknium1
2026-09-20 00:00:24 -07:00
committed by Teknium
parent b3d7c06fe4
commit f3d98d4fda
10 changed files with 74 additions and 26 deletions

View File

@@ -184,9 +184,11 @@ export function transcribeAudio(dataUrl: string, mimeType?: string): Promise<Aud
})
}
export function speakText(text: string): Promise<AudioSpeakResponse> {
// `profile` = the speaking session's owner profile (a Bot's own TTS voice);
// omitted → the active profile.
export function speakText(text: string, profile?: null | string): Promise<AudioSpeakResponse> {
return hermesApi<AudioSpeakResponse>({
...profileScoped(),
...profileScoped(profile || undefined),
path: '/api/audio/speak',
method: 'POST',
body: { text },

View File

@@ -43,9 +43,9 @@ export function useAutoSpeakReplies({
const enabled = useStore($autoSpeakReplies)
// Wake on THIS composer's transcript: a tile subscribed to the primary's
// would never fire on its own replies (and would fire on someone else's).
const { $messages } = useComposerScope()
const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply })
latest.current = { conversationActive, failureLabel, markSpoken, pendingReply }
const { $messages, profile } = useComposerScope()
const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply, profile })
latest.current = { conversationActive, failureLabel, markSpoken, pendingReply, profile }
useEffect(() => {
if (!enabled) {
@@ -57,7 +57,7 @@ export function useAutoSpeakReplies({
latest.current.markSpoken()
const speakLatest = () => {
const { conversationActive, failureLabel, markSpoken, pendingReply } = latest.current
const { conversationActive, failureLabel, markSpoken, pendingReply, profile } = latest.current
if (conversationActive || $voicePlayback.get().status !== 'idle') {
return
@@ -75,7 +75,7 @@ export function useAutoSpeakReplies({
// ran in every window, so peers just stay quiet.
void ownsAmbientCue(`speak:${reply.id}`).then(owns => {
if (owns) {
void playSpeechText(reply.text, { messageId: reply.id, source: 'read-aloud' }).catch(error =>
void playSpeechText(reply.text, { messageId: reply.id, profile, source: 'read-aloud' }).catch(error =>
notifyError(error, failureLabel)
)
}

View File

@@ -14,6 +14,8 @@ import { isVoiceStopCommand } from '@/lib/voice-stop-word'
import { notify, notifyError } from '@/store/notifications'
import { $voicePlayback } from '@/store/voice-playback'
import { useComposerScope } from '../scope'
import { useMicRecorder } from './use-mic-recorder'
export type ConversationStatus = 'idle' | 'listening' | 'transcribing' | 'thinking' | 'speaking'
@@ -60,6 +62,11 @@ export function useVoiceConversation({
const { t } = useI18n()
const voiceCopy = t.notifications.voice
const { handle, level } = useMicRecorder(voiceCopy)
// The scope's session owner (a Bot's own profile) picks the TTS voice; a
// ref keeps the long-lived turn closures below reading the current value.
const { profile: ownerProfile } = useComposerScope()
const ownerProfileRef = useRef(ownerProfile)
ownerProfileRef.current = ownerProfile
const [status, setStatus] = useState<ConversationStatus>('idle')
const [muted, setMuted] = useState(false)
const turnTimeoutRef = useRef<number | null>(null)
@@ -459,7 +466,10 @@ export function useVoiceConversation({
// this is a safety net for read-aloud-style entries into the loop.
ensureBargeMonitor()
const playback = playSpeechText(response.text, { source: 'voice-conversation' })
const playback = playSpeechText(response.text, {
profile: ownerProfileRef.current,
source: 'voice-conversation'
})
// playSpeechText performs its normal cleanup synchronously before
// returning. Capture the sequence after that internal increment so
// only a later, external stop suppresses the next listen cycle.
@@ -500,7 +510,7 @@ export function useVoiceConversation({
ensureBargeMonitor()
void (async () => {
const session = await startSpeechStream({ source: 'voice-conversation' })
const session = await startSpeechStream({ profile: ownerProfileRef.current, source: 'voice-conversation' })
// The session may resolve after the loop moved on (barge, disable).
if (responseIdRef.current !== responseId) {

View File

@@ -27,6 +27,9 @@ export interface ComposerScope {
* keep streaming out of the composer's renders; subscribe only off-render
* (auto-speak) where the reply edge is the whole point. */
$messages: ReadableAtom<ChatMessage[]>
/** Owner profile of this scope's session (a Bot tile runs on the Bot's own
* profile). Voice playback synthesizes with it; undefined → active profile. */
profile?: null | string
/** Focus-bus routing key (`'main'` | `'tile:<id>'`). */
target: ComposerTarget
}

View File

@@ -226,9 +226,10 @@ function TileChat({
$awaitingInput: sessionAwaitingInput(runtimeId),
$messages: view.$messages,
attachments,
profile: ownerRoute?.targetProfile || ownerRoute?.profile || undefined,
target: `tile:${storedSessionId}`
}),
[attachments, runtimeId, storedSessionId, view.$messages]
[attachments, ownerRoute?.profile, ownerRoute?.targetProfile, runtimeId, storedSessionId, view.$messages]
)
// Tile actions must keep the persisted owner route. The ambient gateway hook

View File

@@ -13,6 +13,7 @@ import { type FC, type ReactNode, useCallback, useContext, useEffect, useMemo, u
import { useInRouterContext, useNavigate } from 'react-router'
import { requestModelMenuToggle } from '@/app/chat/composer/focus'
import { useComposerScope } from '@/app/chat/composer/scope'
import { useSessionView } from '@/app/chat/session-view'
import { SETTINGS_ROUTE } from '@/app/routes'
import { dispatchedTo } from '@/components/assistant-ui/thread/agent-delivery'
@@ -1043,6 +1044,8 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get
const voicePlayback = useStore($voicePlayback)
const view = useSessionView()
const sessionId = useStore(view.$runtimeId)
// A Bot tile's session owns its own profile → its own TTS voice.
const { profile } = useComposerScope()
const readAloudStatus =
voicePlayback.source === 'read-aloud' && voicePlayback.messageId === messageId ? voicePlayback.status : 'idle'
@@ -1061,12 +1064,12 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get
}
try {
await playSpeechText(text, { messageId, source: 'read-aloud' })
await playSpeechText(text, { messageId, profile, source: 'read-aloud' })
markAssistantIdSpoken(sessionId, view.$messages.get(), messageId)
} catch (error) {
notifyError(error, copy.readAloudFailed)
}
}, [copy.readAloudFailed, getText, messageId, sessionId, view.$messages])
}, [copy.readAloudFailed, getText, messageId, profile, sessionId, view.$messages])
return (
<TooltipIconButton

View File

@@ -77,6 +77,21 @@ describe('fetchVoiceClientConfig', () => {
expect(api).toHaveBeenCalledTimes(2)
})
// A Bot chat runs on the Bot's own profile: two Bots open beside the active
// profile must each fetch THEIR profile's TTS config, and a plain chat keeps
// the active-profile fallback (#100864).
it('resolves per-Bot TTS config by the owner profile, keeping the active-profile fallback', async () => {
const api = mockDesktopApi({ ok: true, stt: directStt, tts: relay })
setApiRequestProfile('research')
await fetchVoiceClientConfig('bot-rachel')
await fetchVoiceClientConfig('bot-adam')
await fetchVoiceClientConfig()
const profiles = api.mock.calls.map(([request]) => (request as { profile?: string }).profile)
expect(profiles).toEqual(['bot-rachel', 'bot-adam', 'research'])
})
it('resolves null on an older backend without the endpoint', async () => {
Object.defineProperty(window, 'hermesDesktop', {
configurable: true,

View File

@@ -70,8 +70,10 @@ const STT_REQUEST_TIMEOUT_MS = 60_000
let cached: { key: string; at: number; config: VoiceClientConfig } | null = null
let inflight: { key: string; promise: Promise<null | VoiceClientConfig> } | null = null
function scopeKey(): string {
return `${getApiRequestConnection() ?? 'local'}::${getApiRequestProfile() ?? 'default'}`
// `profile` is the session OWNER's profile (a Bot chat runs on its own
// profile with its own TTS voice); undefined/null → the active profile.
function scopeKey(profile?: null | string): string {
return `${getApiRequestConnection() ?? 'local'}::${profile || getApiRequestProfile() || 'default'}`
}
/** Drop cached credentials (used by tests; scope changes rotate the key). */
@@ -80,8 +82,8 @@ export function clearVoiceClientConfigCache(): void {
inflight = null
}
export async function fetchVoiceClientConfig(): Promise<null | VoiceClientConfig> {
const key = scopeKey()
export async function fetchVoiceClientConfig(profile?: null | string): Promise<null | VoiceClientConfig> {
const key = scopeKey(profile)
if (cached && cached.key === key && Date.now() - cached.at < CONFIG_TTL_MS) {
return cached.config
@@ -97,7 +99,7 @@ export async function fetchVoiceClientConfig(): Promise<null | VoiceClientConfig
// profile — the same routing every relay audio call uses, so the
// config comes from the backend the user is actually talking to.
const response = await hermesApi<{ ok: boolean } & VoiceClientConfig>({
...profileScoped(),
...profileScoped(profile || undefined),
path: '/api/audio/voice-config'
})
@@ -316,8 +318,8 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise<null | s
// ---------------------------------------------------------------------------
/** Resolve the profile's TTS config when it is client-callable, else null. */
export async function directTtsConfig(): Promise<DirectTtsConfig | null> {
const config = await fetchVoiceClientConfig()
export async function directTtsConfig(profile?: null | string): Promise<DirectTtsConfig | null> {
const config = await fetchVoiceClientConfig(profile)
return config?.tts && config.tts.mode === 'direct' ? config.tts : null
}

View File

@@ -71,6 +71,15 @@ describe('resolveSpeakStreamUrl', () => {
expect(getConnectionFor).not.toHaveBeenCalled()
})
it("dials the speaking session's owner profile ahead of the active profile", async () => {
setApiRequestProfile('research')
const url = await resolveSpeakStreamUrl('bot-adam')
expect(url).toContain('profile=bot-adam')
expect(getConnection).toHaveBeenCalledWith('bot-adam')
})
it('preserves a backend-namespace profile already minted into the ws URL', async () => {
// SSH remoteProfile aliasing / sharedRemote scoping: the registry mint
// writes the BACKEND's profile name into the URL. The desktop-side

View File

@@ -73,6 +73,9 @@ function currentState(
export interface VoicePlaybackOptions {
messageId?: string | null
/** Owner profile of the speaking session (a Bot chat synthesizes with its
* own profile's TTS voice). Omitted → the active profile. */
profile?: null | string
source: VoicePlaybackSource
}
@@ -105,7 +108,7 @@ export function stopVoicePlayback() {
/** Exported for tests: the (connection, profile) routing contract below is
* exactly what broke in the desktop-remote voice report — keep it pinned. */
export async function resolveSpeakStreamUrl(): Promise<null | string> {
export async function resolveSpeakStreamUrl(ownerProfile?: null | string): Promise<null | string> {
const desktop = window.hermesDesktop
if (!desktop?.getConnection) {
@@ -123,7 +126,7 @@ export async function resolveSpeakStreamUrl(): Promise<null | string> {
// replies would synthesize with the local (often unconfigured) TTS
// instead of the profile the user is actually talking to (#90051-adjacent
// desktop-remote voice report, Aug 2026).
const profile = getApiRequestProfile()
const profile = ownerProfile || getApiRequestProfile()
const connectionId = getApiRequestConnection()
// Both awaits below are IPC round-trips into the main process with no
@@ -522,7 +525,7 @@ function openSpeechStream(wsUrl: string, options: VoicePlaybackOptions): SpeechS
* `playSpeechText`).
*/
export async function startSpeechStream(options: VoicePlaybackOptions): Promise<null | SpeechStreamSession> {
const direct = await directTtsConfig().catch(() => null)
const direct = await directTtsConfig(options.profile).catch(() => null)
if (direct) {
stopVoicePlayback()
@@ -539,7 +542,7 @@ export async function startSpeechStream(options: VoicePlaybackOptions): Promise<
return session
}
const wsUrl = await resolveSpeakStreamUrl()
const wsUrl = await resolveSpeakStreamUrl(options.profile)
if (!wsUrl) {
return null
@@ -573,7 +576,7 @@ async function playSpeechDataUrl(
options: VoicePlaybackOptions,
isCurrent: () => boolean
): Promise<boolean> {
const response = await speakText(speakableText)
const response = await speakText(speakableText, options.profile)
if (!isCurrent()) {
return false
@@ -669,7 +672,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions
try {
// Ladder: client-direct synthesis (profile's own TTS, no gateway audio
// hop) → streaming WS relay → POST data-URL fallback.
const direct = await directTtsConfig().catch(() => null)
const direct = await directTtsConfig(options.profile).catch(() => null)
if (direct && isCurrent()) {
const session = openClientDirectSpeechSession(direct, options)
@@ -693,7 +696,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions
return false
}
const streamUrl = await resolveSpeakStreamUrl()
const streamUrl = await resolveSpeakStreamUrl(options.profile)
if (streamUrl && isCurrent()) {
const outcome = await playSpeechStream(streamUrl, speakableText, options)