fix(desktop): Bot chats speak through their own (connection, profile) in tiles and the main pane
Review follow-up on the Bot voice routing (#100864, salvage #101545 @FalconOrtiz). - Owner identity is (connection, profile) per the Bot Mode standing ruling: ComposerScope publishes `connectionId` beside `profile`; `voice-client-direct` keys its config cache and `hermesApi` scope on both (new `ownerScoped` in api/client.ts, beside `profileScoped`), and `voice-playback` mints the speak-stream against the OWNER connection (`getConnectionFor`) and pins the POST fallback the same way. Two `default` Bots on two gateways no longer collide in the cache or mint against the active gateway. - Main pane: `ChatRuntimeBoundary` publishes the session owner hint's (connection, profile) on the composer scope when the ambient scope has no owner, so a Bot chat opened in place (openStoredBotChat) speaks with its owner voice; a tile's scope already names its owner and is kept as is. - Tests: the two direct helper tests are folded into (1) a production-path test that renders useAutoSpeakReplies under a Bot scope and asserts every REST audio leg carries the owner (red when the hook's wiring is reverted to base AND when only `profile` is threaded), and (2) the routing test on resolveSpeakStreamUrl with an owner object. - Docs: the fallback sentence matches the code (profile server defaults, active profile only for ownerless chats); STT stays on the active profile. Co-authored-by: FalconOrtiz <falcon.ortiz11@gmail.com>
This commit is contained in:
@@ -72,6 +72,23 @@ export function profileScoped(profile?: null | string): { priority?: 'foreground
|
||||
}
|
||||
}
|
||||
|
||||
/** A session's OWNER as a request scope: a profile belongs to ONE gateway, so
|
||||
* a Bot on another connection is (its connection, its profile) — never the
|
||||
* active connection with the Bot's profile name. Missing halves fall back to
|
||||
* the ambient scope; an explicit connection — `'local'` included — overrides
|
||||
* the ambient tag `hermesApi` spreads underneath (as capabilityScoped does). */
|
||||
export interface OwnerScope {
|
||||
connectionId?: null | string
|
||||
profile?: null | string
|
||||
}
|
||||
|
||||
export function ownerScoped(owner?: OwnerScope): { connectionId?: string; priority?: 'foreground'; profile?: string } {
|
||||
return {
|
||||
...profileScoped(owner?.profile || undefined),
|
||||
...(owner?.connectionId ? { connectionId: owner.connectionId } : {})
|
||||
}
|
||||
}
|
||||
|
||||
/** Profile that profile-scoped REST/WS calls should target (null → primary).
|
||||
* Read-only twin of setApiRequestProfile for modules (e.g. voice playback)
|
||||
* that build their own connection URLs and must stay on the same backend. */
|
||||
|
||||
@@ -13,7 +13,7 @@ import type {
|
||||
MemoryStatusResponse
|
||||
} from '@/types/hermes'
|
||||
|
||||
import { capabilityScoped, hermesApi, type ProfileScope, profileScoped } from './client'
|
||||
import { capabilityScoped, hermesApi, type OwnerScope, ownerScoped, type ProfileScope, profileScoped } from './client'
|
||||
|
||||
export const AUDIO_SPEAK_MIN_REQUEST_TIMEOUT_MS = 180_000
|
||||
export const AUDIO_SPEAK_MAX_REQUEST_TIMEOUT_MS = 600_000
|
||||
@@ -184,11 +184,11 @@ export function transcribeAudio(dataUrl: string, mimeType?: string): Promise<Aud
|
||||
})
|
||||
}
|
||||
|
||||
// `profile` = the speaking session's owner profile (a Bot's own TTS voice);
|
||||
// omitted → the active profile.
|
||||
export function speakText(text: string, profile?: null | string): Promise<AudioSpeakResponse> {
|
||||
// `owner` = the speaking session's (connection, profile) — a Bot's own TTS
|
||||
// voice on its own gateway; omitted halves → the active scope.
|
||||
export function speakText(text: string, owner?: OwnerScope): Promise<AudioSpeakResponse> {
|
||||
return hermesApi<AudioSpeakResponse>({
|
||||
...profileScoped(profile || undefined),
|
||||
...ownerScoped(owner),
|
||||
path: '/api/audio/speak',
|
||||
method: 'POST',
|
||||
body: { text },
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
import { cleanup, renderHook } from '@testing-library/react'
|
||||
import { atom } from 'nanostores'
|
||||
import type { ReactNode } from 'react'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { setApiRequestConnection, setApiRequestProfile } from '@/hermes'
|
||||
import { clearVoiceClientConfigCache } from '@/lib/voice-client-direct'
|
||||
import { $autoSpeakReplies } from '@/store/voice-prefs'
|
||||
|
||||
import { ComposerScopeProvider, MAIN_COMPOSER_SCOPE } from '../scope'
|
||||
|
||||
import { useAutoSpeakReplies } from './use-auto-speak-replies'
|
||||
|
||||
vi.mock('@/store/ambient', () => ({ ownsAmbientCue: async () => true }))
|
||||
vi.mock('@/store/notifications', () => ({ notifyError: vi.fn() }))
|
||||
|
||||
// A Bot chat is owned by (its connection, its profile). The production path —
|
||||
// the auto-speak hook reading its composer scope, through playSpeechText's
|
||||
// ladder, down to the REST audio calls — must carry that owner, or the Bot
|
||||
// speaks with the active profile's voice on the active gateway (#100864).
|
||||
describe('useAutoSpeakReplies — owner-routed synthesis', () => {
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
$autoSpeakReplies.set(false)
|
||||
setApiRequestConnection(null)
|
||||
setApiRequestProfile(null)
|
||||
clearVoiceClientConfigCache()
|
||||
Reflect.deleteProperty(window, 'hermesDesktop')
|
||||
})
|
||||
|
||||
it('synthesizes a Bot reply with the scope owner (connection, profile), not the active scope', async () => {
|
||||
const api = vi.fn(async ({ path }: { path: string }) =>
|
||||
path.startsWith('/api/audio/voice-config') ? { ok: false } : { audio: '' }
|
||||
)
|
||||
|
||||
Object.defineProperty(window, 'hermesDesktop', { configurable: true, value: { api } })
|
||||
setApiRequestConnection('gw-active')
|
||||
setApiRequestProfile('research')
|
||||
$autoSpeakReplies.set(true)
|
||||
|
||||
const $messages = atom<never[]>([])
|
||||
let reply: null | { id: string; pending: boolean; text: string } = null
|
||||
|
||||
const wrapper = ({ children }: { children: ReactNode }) => (
|
||||
<ComposerScopeProvider
|
||||
value={{ ...MAIN_COMPOSER_SCOPE, $messages, connectionId: 'gw-bots', profile: 'bot-adam', target: 'tile:bot' }}
|
||||
>
|
||||
{children}
|
||||
</ComposerScopeProvider>
|
||||
)
|
||||
|
||||
renderHook(
|
||||
() =>
|
||||
useAutoSpeakReplies({
|
||||
conversationActive: false,
|
||||
failureLabel: 'failed',
|
||||
markSpoken: () => {
|
||||
reply = null
|
||||
},
|
||||
pendingReply: () => reply,
|
||||
sessionId: 'bot-session'
|
||||
}),
|
||||
{ wrapper }
|
||||
)
|
||||
|
||||
reply = { id: 'm1', pending: false, text: 'Hello from Adam.' }
|
||||
$messages.set([])
|
||||
|
||||
await vi.waitFor(() =>
|
||||
expect(api.mock.calls.map(([request]) => (request as { path: string }).path)).toContain('/api/audio/speak')
|
||||
)
|
||||
|
||||
const scopes = new Set(
|
||||
api.mock.calls.map(([request]) => {
|
||||
const { connectionId, profile } = request as { connectionId?: string; profile?: string }
|
||||
|
||||
return `${connectionId}::${profile}`
|
||||
})
|
||||
)
|
||||
|
||||
// voice-config AND the speak POST — every leg names the Bot's owner.
|
||||
expect(scopes).toEqual(new Set(['gw-bots::bot-adam']))
|
||||
})
|
||||
})
|
||||
@@ -43,9 +43,9 @@ export function useAutoSpeakReplies({
|
||||
const enabled = useStore($autoSpeakReplies)
|
||||
// Wake on THIS composer's transcript: a tile subscribed to the primary's
|
||||
// would never fire on its own replies (and would fire on someone else's).
|
||||
const { $messages, profile } = useComposerScope()
|
||||
const latest = useRef({ conversationActive, failureLabel, markSpoken, pendingReply, profile })
|
||||
latest.current = { conversationActive, failureLabel, markSpoken, pendingReply, profile }
|
||||
const { $messages, connectionId, profile } = useComposerScope()
|
||||
const latest = useRef({ connectionId, conversationActive, failureLabel, markSpoken, pendingReply, profile })
|
||||
latest.current = { connectionId, conversationActive, failureLabel, markSpoken, pendingReply, profile }
|
||||
|
||||
useEffect(() => {
|
||||
if (!enabled) {
|
||||
@@ -57,7 +57,7 @@ export function useAutoSpeakReplies({
|
||||
latest.current.markSpoken()
|
||||
|
||||
const speakLatest = () => {
|
||||
const { conversationActive, failureLabel, markSpoken, pendingReply, profile } = latest.current
|
||||
const { connectionId, conversationActive, failureLabel, markSpoken, pendingReply, profile } = latest.current
|
||||
|
||||
if (conversationActive || $voicePlayback.get().status !== 'idle') {
|
||||
return
|
||||
@@ -75,8 +75,8 @@ export function useAutoSpeakReplies({
|
||||
// ran in every window, so peers just stay quiet.
|
||||
void ownsAmbientCue(`speak:${reply.id}`).then(owns => {
|
||||
if (owns) {
|
||||
void playSpeechText(reply.text, { messageId: reply.id, profile, source: 'read-aloud' }).catch(error =>
|
||||
notifyError(error, failureLabel)
|
||||
void playSpeechText(reply.text, { connectionId, messageId: reply.id, profile, source: 'read-aloud' }).catch(
|
||||
error => notifyError(error, failureLabel)
|
||||
)
|
||||
}
|
||||
})
|
||||
|
||||
@@ -62,11 +62,12 @@ export function useVoiceConversation({
|
||||
const { t } = useI18n()
|
||||
const voiceCopy = t.notifications.voice
|
||||
const { handle, level } = useMicRecorder(voiceCopy)
|
||||
// The scope's session owner (a Bot's own profile) picks the TTS voice; a
|
||||
// ref keeps the long-lived turn closures below reading the current value.
|
||||
const { profile: ownerProfile } = useComposerScope()
|
||||
const ownerProfileRef = useRef(ownerProfile)
|
||||
ownerProfileRef.current = ownerProfile
|
||||
// The scope's session owner (a Bot's own connection + profile) picks the TTS
|
||||
// voice; a ref keeps the long-lived turn closures below reading the current
|
||||
// value.
|
||||
const { connectionId: ownerConnectionId, profile: ownerProfile } = useComposerScope()
|
||||
const ownerRef = useRef({ connectionId: ownerConnectionId, profile: ownerProfile })
|
||||
ownerRef.current = { connectionId: ownerConnectionId, profile: ownerProfile }
|
||||
const [status, setStatus] = useState<ConversationStatus>('idle')
|
||||
const [muted, setMuted] = useState(false)
|
||||
const turnTimeoutRef = useRef<number | null>(null)
|
||||
@@ -466,10 +467,7 @@ export function useVoiceConversation({
|
||||
// this is a safety net for read-aloud-style entries into the loop.
|
||||
ensureBargeMonitor()
|
||||
|
||||
const playback = playSpeechText(response.text, {
|
||||
profile: ownerProfileRef.current,
|
||||
source: 'voice-conversation'
|
||||
})
|
||||
const playback = playSpeechText(response.text, { ...ownerRef.current, source: 'voice-conversation' })
|
||||
// playSpeechText performs its normal cleanup synchronously before
|
||||
// returning. Capture the sequence after that internal increment so
|
||||
// only a later, external stop suppresses the next listen cycle.
|
||||
@@ -510,7 +508,7 @@ export function useVoiceConversation({
|
||||
ensureBargeMonitor()
|
||||
|
||||
void (async () => {
|
||||
const session = await startSpeechStream({ profile: ownerProfileRef.current, source: 'voice-conversation' })
|
||||
const session = await startSpeechStream({ ...ownerRef.current, source: 'voice-conversation' })
|
||||
|
||||
// The session may resolve after the loop moved on (barge, disable).
|
||||
if (responseIdRef.current !== responseId) {
|
||||
|
||||
@@ -27,6 +27,10 @@ export interface ComposerScope {
|
||||
* keep streaming out of the composer's renders; subscribe only off-render
|
||||
* (auto-speak) where the reply edge is the whole point. */
|
||||
$messages: ReadableAtom<ChatMessage[]>
|
||||
/** Owner connection of this scope's session — a profile belongs to ONE
|
||||
* gateway, so voice playback mints against it (cross-connection Bots).
|
||||
* undefined → the active connection. */
|
||||
connectionId?: null | string
|
||||
/** Owner profile of this scope's session (a Bot tile runs on the Bot's own
|
||||
* profile). Voice playback synthesizes with it; undefined → active profile. */
|
||||
profile?: null | string
|
||||
|
||||
@@ -60,7 +60,7 @@ import { ChatBar, ChatBarFallback } from './composer'
|
||||
import { FloatingComposerSurface } from './composer/floating-surface'
|
||||
import { requestComposerInsert } from './composer/focus'
|
||||
import { droppedFileInlineRefs } from './composer/inline-refs'
|
||||
import { ComposerSurfaceProvider, useComposerScope, useComposerSurfaceId } from './composer/scope'
|
||||
import { ComposerScopeProvider, ComposerSurfaceProvider, useComposerScope, useComposerSurfaceId } from './composer/scope'
|
||||
import type { ChatBarState } from './composer/types'
|
||||
import { useHistoryWindow } from './history-window'
|
||||
import { type DroppedFile, partitionDroppedFiles } from './hooks/use-composer-actions'
|
||||
@@ -261,6 +261,20 @@ export function ChatRuntimeBoundary({
|
||||
? { connectionId: ownerConnection, profile: ownerProfile }
|
||||
: undefined, [ownerConnection, ownerProfile])
|
||||
|
||||
// A Bot chat opened IN PLACE in the main pane (openStoredBotChat) keeps the
|
||||
// active profile, so the ambient scope carries no owner. Publish the session
|
||||
// owner hint's (connection, profile) here so voice playback speaks with the
|
||||
// Bot's own voice; a tile's scope already names its owner and is kept as is.
|
||||
const parentScope = useComposerScope()
|
||||
|
||||
const composerScope = useMemo(
|
||||
() =>
|
||||
parentScope.profile || !ownerProfile
|
||||
? parentScope
|
||||
: { ...parentScope, connectionId: ownerConnection || undefined, profile: ownerProfile },
|
||||
[ownerConnection, ownerProfile, parentScope]
|
||||
)
|
||||
|
||||
const history = useHistoryWindow({
|
||||
scopeKey: JSON.stringify([runtimeId, storedId, tailProfile, connectionId, activeProfile, suppressMessages]),
|
||||
storedId,
|
||||
@@ -399,9 +413,11 @@ export function ChatRuntimeBoundary({
|
||||
})
|
||||
|
||||
return (
|
||||
<TranscriptWindowProvider value={transcriptWindow}>
|
||||
<AssistantRuntimeProvider runtime={runtime}>{children}</AssistantRuntimeProvider>
|
||||
</TranscriptWindowProvider>
|
||||
<ComposerScopeProvider value={composerScope}>
|
||||
<TranscriptWindowProvider value={transcriptWindow}>
|
||||
<AssistantRuntimeProvider runtime={runtime}>{children}</AssistantRuntimeProvider>
|
||||
</TranscriptWindowProvider>
|
||||
</ComposerScopeProvider>
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -226,10 +226,19 @@ function TileChat({
|
||||
$awaitingInput: sessionAwaitingInput(runtimeId),
|
||||
$messages: view.$messages,
|
||||
attachments,
|
||||
connectionId: ownerRoute?.connectionId || undefined,
|
||||
profile: ownerRoute?.targetProfile || ownerRoute?.profile || undefined,
|
||||
target: `tile:${storedSessionId}`
|
||||
}),
|
||||
[attachments, ownerRoute?.profile, ownerRoute?.targetProfile, runtimeId, storedSessionId, view.$messages]
|
||||
[
|
||||
attachments,
|
||||
ownerRoute?.connectionId,
|
||||
ownerRoute?.profile,
|
||||
ownerRoute?.targetProfile,
|
||||
runtimeId,
|
||||
storedSessionId,
|
||||
view.$messages
|
||||
]
|
||||
)
|
||||
|
||||
// Tile actions must keep the persisted owner route. The ambient gateway hook
|
||||
|
||||
@@ -710,6 +710,7 @@ const ScheduledRetryAction: FC<{ resetsAt: number }> = ({ resetsAt }) => {
|
||||
},
|
||||
Math.max(0, fireAt - Date.now())
|
||||
)
|
||||
|
||||
const tick = window.setInterval(() => setNow(Date.now()), 1000)
|
||||
|
||||
return () => {
|
||||
@@ -1044,8 +1045,8 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get
|
||||
const voicePlayback = useStore($voicePlayback)
|
||||
const view = useSessionView()
|
||||
const sessionId = useStore(view.$runtimeId)
|
||||
// A Bot tile's session owns its own profile → its own TTS voice.
|
||||
const { profile } = useComposerScope()
|
||||
// A Bot chat's session owns its own (connection, profile) → its own TTS voice.
|
||||
const { connectionId, profile } = useComposerScope()
|
||||
|
||||
const readAloudStatus =
|
||||
voicePlayback.source === 'read-aloud' && voicePlayback.messageId === messageId ? voicePlayback.status : 'idle'
|
||||
@@ -1064,12 +1065,12 @@ const ReadAloudButton: FC<{ getText: () => string; messageId: string }> = ({ get
|
||||
}
|
||||
|
||||
try {
|
||||
await playSpeechText(text, { messageId, profile, source: 'read-aloud' })
|
||||
await playSpeechText(text, { connectionId, messageId, profile, source: 'read-aloud' })
|
||||
markAssistantIdSpoken(sessionId, view.$messages.get(), messageId)
|
||||
} catch (error) {
|
||||
notifyError(error, copy.readAloudFailed)
|
||||
}
|
||||
}, [copy.readAloudFailed, getText, messageId, profile, sessionId, view.$messages])
|
||||
}, [connectionId, copy.readAloudFailed, getText, messageId, profile, sessionId, view.$messages])
|
||||
|
||||
return (
|
||||
<TooltipIconButton
|
||||
|
||||
@@ -77,21 +77,6 @@ describe('fetchVoiceClientConfig', () => {
|
||||
expect(api).toHaveBeenCalledTimes(2)
|
||||
})
|
||||
|
||||
// A Bot chat runs on the Bot's own profile: two Bots open beside the active
|
||||
// profile must each fetch THEIR profile's TTS config, and a plain chat keeps
|
||||
// the active-profile fallback (#100864).
|
||||
it('resolves per-Bot TTS config by the owner profile, keeping the active-profile fallback', async () => {
|
||||
const api = mockDesktopApi({ ok: true, stt: directStt, tts: relay })
|
||||
setApiRequestProfile('research')
|
||||
|
||||
await fetchVoiceClientConfig('bot-rachel')
|
||||
await fetchVoiceClientConfig('bot-adam')
|
||||
await fetchVoiceClientConfig()
|
||||
|
||||
const profiles = api.mock.calls.map(([request]) => (request as { profile?: string }).profile)
|
||||
expect(profiles).toEqual(['bot-rachel', 'bot-adam', 'research'])
|
||||
})
|
||||
|
||||
it('resolves null on an older backend without the endpoint', async () => {
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { profileScoped } from '@/api/client'
|
||||
import { type OwnerScope, ownerScoped } from '@/api/client'
|
||||
import { getApiRequestConnection, getApiRequestProfile, hermesApi } from '@/hermes'
|
||||
|
||||
/**
|
||||
@@ -70,10 +70,10 @@ const STT_REQUEST_TIMEOUT_MS = 60_000
|
||||
let cached: { key: string; at: number; config: VoiceClientConfig } | null = null
|
||||
let inflight: { key: string; promise: Promise<null | VoiceClientConfig> } | null = null
|
||||
|
||||
// `profile` is the session OWNER's profile (a Bot chat runs on its own
|
||||
// profile with its own TTS voice); undefined/null → the active profile.
|
||||
function scopeKey(profile?: null | string): string {
|
||||
return `${getApiRequestConnection() ?? 'local'}::${profile || getApiRequestProfile() || 'default'}`
|
||||
// `owner` is the speaking session's (connection, profile) — a Bot chat runs
|
||||
// on its own profile, on its own gateway; missing halves → the active scope.
|
||||
function scopeKey(owner?: OwnerScope): string {
|
||||
return `${owner?.connectionId || getApiRequestConnection() || 'local'}::${owner?.profile || getApiRequestProfile() || 'default'}`
|
||||
}
|
||||
|
||||
/** Drop cached credentials (used by tests; scope changes rotate the key). */
|
||||
@@ -82,8 +82,8 @@ export function clearVoiceClientConfigCache(): void {
|
||||
inflight = null
|
||||
}
|
||||
|
||||
export async function fetchVoiceClientConfig(profile?: null | string): Promise<null | VoiceClientConfig> {
|
||||
const key = scopeKey(profile)
|
||||
export async function fetchVoiceClientConfig(owner?: OwnerScope): Promise<null | VoiceClientConfig> {
|
||||
const key = scopeKey(owner)
|
||||
|
||||
if (cached && cached.key === key && Date.now() - cached.at < CONFIG_TTL_MS) {
|
||||
return cached.config
|
||||
@@ -99,7 +99,7 @@ export async function fetchVoiceClientConfig(profile?: null | string): Promise<n
|
||||
// profile — the same routing every relay audio call uses, so the
|
||||
// config comes from the backend the user is actually talking to.
|
||||
const response = await hermesApi<{ ok: boolean } & VoiceClientConfig>({
|
||||
...profileScoped(profile || undefined),
|
||||
...ownerScoped(owner),
|
||||
path: '/api/audio/voice-config'
|
||||
})
|
||||
|
||||
@@ -318,8 +318,8 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise<null | s
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Resolve the profile's TTS config when it is client-callable, else null. */
|
||||
export async function directTtsConfig(profile?: null | string): Promise<DirectTtsConfig | null> {
|
||||
const config = await fetchVoiceClientConfig(profile)
|
||||
export async function directTtsConfig(owner?: OwnerScope): Promise<DirectTtsConfig | null> {
|
||||
const config = await fetchVoiceClientConfig(owner)
|
||||
|
||||
return config?.tts && config.tts.mode === 'direct' ? config.tts : null
|
||||
}
|
||||
|
||||
@@ -71,13 +71,18 @@ describe('resolveSpeakStreamUrl', () => {
|
||||
expect(getConnectionFor).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it("dials the speaking session's owner profile ahead of the active profile", async () => {
|
||||
// A Bot on another registered gateway is (its connection, its profile): the
|
||||
// stream must mint against the Bot's OWN connection, never the active one
|
||||
// with the Bot's profile name (two `default` Bots on two gateways).
|
||||
it("dials the speaking session's owner (connection, profile) ahead of the active scope", async () => {
|
||||
setApiRequestConnection('gw-active')
|
||||
setApiRequestProfile('research')
|
||||
|
||||
const url = await resolveSpeakStreamUrl('bot-adam')
|
||||
const url = await resolveSpeakStreamUrl({ connectionId: 'gw-bots', profile: 'bot-adam' })
|
||||
|
||||
expect(url).toContain('profile=bot-adam')
|
||||
expect(getConnection).toHaveBeenCalledWith('bot-adam')
|
||||
expect(getConnectionFor).toHaveBeenCalledWith({ connectionId: 'gw-bots', profile: 'bot-adam' })
|
||||
expect(getGatewayWsUrlFor).toHaveBeenCalledWith({ connectionId: 'gw-bots', profile: 'bot-adam' })
|
||||
})
|
||||
|
||||
it('preserves a backend-namespace profile already minted into the ws URL', async () => {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { resolveGatewayWsUrl } from '@hermes/shared'
|
||||
|
||||
import type { OwnerScope } from '@/api/client'
|
||||
import { getApiRequestConnection, getApiRequestProfile, speakText } from '@/hermes'
|
||||
import {
|
||||
cutSentences,
|
||||
@@ -71,11 +72,11 @@ function currentState(
|
||||
}
|
||||
}
|
||||
|
||||
export interface VoicePlaybackOptions {
|
||||
/** The speaking session's owner: a Bot chat synthesizes with its own
|
||||
* profile's TTS voice, minted against the Bot's own connection. Omitted
|
||||
* halves → the active (connection, profile). */
|
||||
export interface VoicePlaybackOptions extends OwnerScope {
|
||||
messageId?: string | null
|
||||
/** Owner profile of the speaking session (a Bot chat synthesizes with its
|
||||
* own profile's TTS voice). Omitted → the active profile. */
|
||||
profile?: null | string
|
||||
source: VoicePlaybackSource
|
||||
}
|
||||
|
||||
@@ -108,7 +109,7 @@ export function stopVoicePlayback() {
|
||||
|
||||
/** Exported for tests: the (connection, profile) routing contract below is
|
||||
* exactly what broke in the desktop-remote voice report — keep it pinned. */
|
||||
export async function resolveSpeakStreamUrl(ownerProfile?: null | string): Promise<null | string> {
|
||||
export async function resolveSpeakStreamUrl(owner?: OwnerScope): Promise<null | string> {
|
||||
const desktop = window.hermesDesktop
|
||||
|
||||
if (!desktop?.getConnection) {
|
||||
@@ -126,8 +127,8 @@ export async function resolveSpeakStreamUrl(ownerProfile?: null | string): Promi
|
||||
// replies would synthesize with the local (often unconfigured) TTS
|
||||
// instead of the profile the user is actually talking to (#90051-adjacent
|
||||
// desktop-remote voice report, Aug 2026).
|
||||
const profile = ownerProfile || getApiRequestProfile()
|
||||
const connectionId = getApiRequestConnection()
|
||||
const profile = owner?.profile || getApiRequestProfile()
|
||||
const connectionId = owner?.connectionId || getApiRequestConnection()
|
||||
|
||||
// Both awaits below are IPC round-trips into the main process with no
|
||||
// timeout of their own (#93454) — a wedged main-process round-trip
|
||||
@@ -525,7 +526,7 @@ function openSpeechStream(wsUrl: string, options: VoicePlaybackOptions): SpeechS
|
||||
* `playSpeechText`).
|
||||
*/
|
||||
export async function startSpeechStream(options: VoicePlaybackOptions): Promise<null | SpeechStreamSession> {
|
||||
const direct = await directTtsConfig(options.profile).catch(() => null)
|
||||
const direct = await directTtsConfig(options).catch(() => null)
|
||||
|
||||
if (direct) {
|
||||
stopVoicePlayback()
|
||||
@@ -542,7 +543,7 @@ export async function startSpeechStream(options: VoicePlaybackOptions): Promise<
|
||||
return session
|
||||
}
|
||||
|
||||
const wsUrl = await resolveSpeakStreamUrl(options.profile)
|
||||
const wsUrl = await resolveSpeakStreamUrl(options)
|
||||
|
||||
if (!wsUrl) {
|
||||
return null
|
||||
@@ -576,7 +577,7 @@ async function playSpeechDataUrl(
|
||||
options: VoicePlaybackOptions,
|
||||
isCurrent: () => boolean
|
||||
): Promise<boolean> {
|
||||
const response = await speakText(speakableText, options.profile)
|
||||
const response = await speakText(speakableText, options)
|
||||
|
||||
if (!isCurrent()) {
|
||||
return false
|
||||
@@ -672,7 +673,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions
|
||||
try {
|
||||
// Ladder: client-direct synthesis (profile's own TTS, no gateway audio
|
||||
// hop) → streaming WS relay → POST data-URL fallback.
|
||||
const direct = await directTtsConfig(options.profile).catch(() => null)
|
||||
const direct = await directTtsConfig(options).catch(() => null)
|
||||
|
||||
if (direct && isCurrent()) {
|
||||
const session = openClientDirectSpeechSession(direct, options)
|
||||
@@ -696,7 +697,7 @@ export async function playSpeechText(text: string, options: VoicePlaybackOptions
|
||||
return false
|
||||
}
|
||||
|
||||
const streamUrl = await resolveSpeakStreamUrl(options.profile)
|
||||
const streamUrl = await resolveSpeakStreamUrl(options)
|
||||
|
||||
if (streamUrl && isCurrent()) {
|
||||
const outcome = await playSpeechStream(streamUrl, speakableText, options)
|
||||
|
||||
@@ -86,7 +86,7 @@ A Bot's look, title, and description are stored in the profile's metadata on the
|
||||
|
||||
## Voices
|
||||
|
||||
A Bot speaks with its **own profile's** TTS settings (`tts.*` in that profile's `config.yaml`). Read Aloud, auto-speak and voice conversation in a Bot chat all synthesize through the Bot's profile, so two Bots with different voices sound different; a Bot whose profile has no TTS config falls back to the active profile's voice.
|
||||
A Bot speaks with its **own profile's** TTS settings (`tts.*` in that profile's `config.yaml`) on its **own gateway**. Read Aloud, auto-speak and voice conversation in a Bot chat — in a split tile or opened in the main pane — all synthesize through the Bot's (connection, profile), so two Bots with different voices sound different and two Bots with the same profile name on different gateways never share a voice. A Bot profile with no `tts.*` of its own uses that profile's server defaults; only a chat with no known owner (a plain session) uses the active profile's voice. Speech-to-text in a Bot chat still uses the active profile's STT settings.
|
||||
|
||||
## Routines
|
||||
|
||||
|
||||
Reference in New Issue
Block a user