fix(desktop): flush direct speech at sealed narration boundaries

This commit is contained in:
funky-xamarin
2026-09-19 05:17:48 +08:00
committed by Teknium
parent bfaa0492d2
commit 96cb6636d3
3 changed files with 152 additions and 2 deletions

View File

@@ -0,0 +1,137 @@
import { act, cleanup, renderHook } from '@testing-library/react'
import { afterEach, expect, it, vi } from 'vitest'
import { type ChatMessage, collectUnspokenTurnSpeech } from '@/lib/chat-messages'
import { stopVoicePlayback } from '@/lib/voice-playback'
import { useVoiceConversation } from './use-voice-conversation'
const mocks = vi.hoisted(() => ({
config: vi.fn(),
mic: {
cancel: vi.fn(),
start: vi.fn(async () => undefined),
stop: vi.fn(async () => ({
audio: new Blob(['fixture']),
heardSpeech: true,
durationMs: 900
}))
}
}))
vi.mock('@/hermes', () => ({
getApiRequestConnection: () => null,
getApiRequestProfile: () => null,
hermesApi: mocks.config,
speakText: vi.fn()
}))
vi.mock('@/api/client', () => ({ profileScoped: (value: unknown) => value }))
vi.mock('./use-mic-recorder', () => ({ useMicRecorder: () => ({ handle: mocks.mic, level: 0 }) }))
vi.mock('@/lib/voice-barge-in', () => ({ monitorSpeechDuringPlayback: () => vi.fn() }))
vi.mock('@/lib/thinking-sound', () => ({ startThinkingSound: vi.fn(), stopThinkingSound: vi.fn() }))
vi.mock('@/store/notifications', () => ({ notify: vi.fn(), notifyError: vi.fn() }))
vi.mock('@/i18n', () => ({ useI18n: () => ({ t: { notifications: { voice: {} } } }) }))
class TestAudio extends EventTarget {
static instances: TestAudio[] = []
src: string
constructor(src: string) {
super()
this.src = src
TestAudio.instances.push(this)
}
play = vi.fn(async () => undefined)
pause = vi.fn()
load = vi.fn()
}
afterEach(() => {
cleanup()
stopVoicePlayback()
vi.useRealTimers()
vi.unstubAllGlobals()
vi.restoreAllMocks()
})
it('speaks a sealed narration while busy and keeps the session open for the final reply', async () => {
vi.useFakeTimers()
TestAudio.instances = []
vi.stubGlobal('Audio', TestAudio)
vi.stubGlobal(
'URL',
class extends URL {
static createObjectURL = vi.fn(() => 'blob:fixture')
static revokeObjectURL = vi.fn()
}
)
mocks.config.mockResolvedValue({
ok: true,
stt: { mode: 'relay' },
tts: {
mode: 'direct',
wire: 'openai-speech',
provider: 'openai',
base_url: 'https://tts.invalid/v1',
api_key: 'fixture-only',
model: 'tts-fixture',
voice: 'fixture',
speed: null
}
})
const inputs: string[] = []
vi.stubGlobal(
'fetch',
vi.fn(async (_url: string, options: RequestInit) => {
inputs.push(JSON.parse(options.body as string).input)
return { ok: true, arrayBuffer: async () => new Uint8Array([1]).buffer }
})
)
const narration = 'Let me check the live state of the branch.'
const answer = 'The branch is clean and the check is complete.'
const messages: ChatMessage[] = []
const hook = renderHook(
({ busy }) =>
useVoiceConversation({
busy,
enabled: true,
consumePendingResponse: vi.fn(),
onSubmit: vi.fn(async () => {
hook.rerender({ busy: true })
}),
onTranscribeAudio: async () => 'Check the branch',
pendingResponse: () => collectUnspokenTurnSpeech(messages, null)
}),
{ initialProps: { busy: false } }
)
await act(async () => {
await hook.result.current.start()
})
await act(async () => {
hook.result.current.stopTurn()
})
messages.push({ id: 'narration', role: 'assistant', pending: true, parts: [{ type: 'text', text: narration }] })
hook.rerender({ busy: true })
await act(async () => {
await vi.advanceTimersByTimeAsync(300)
})
expect(inputs).toEqual([])
messages[0].pending = false
await act(async () => {
await vi.advanceTimersByTimeAsync(300)
})
expect(inputs).toEqual([narration])
expect(TestAudio.instances[0].play).toHaveBeenCalledOnce()
await act(async () => {
TestAudio.instances[0].dispatchEvent(new Event('ended'))
})
expect(mocks.mic.start).toHaveBeenCalledTimes(1)
messages.push({ id: 'answer', role: 'assistant', pending: false, parts: [{ type: 'text', text: answer }] })
hook.rerender({ busy: false })
await act(async () => {
await vi.advanceTimersByTimeAsync(300)
})
expect(inputs).toEqual([narration, answer])
})

View File

@@ -430,8 +430,14 @@ export function useVoiceConversation({
spokenSourceLengthRef.current = response.text.length
}
if (!response.pending && !busyRef.current) {
session.finish()
if (!response.pending) {
// A sealed interim is a committed boundary even while its tool runs.
// Keep the session open for the next bubble, but speak this tail now.
if (busyRef.current) {
session.flush?.()
} else {
session.finish()
}
}
} else if (!busyRef.current) {
// Reply consumed/vanished while we were speaking — close out the turn.

View File

@@ -187,6 +187,8 @@ export async function resolveSpeakStreamUrl(owner?: OwnerScope): Promise<null |
export interface SpeechStreamSession {
/** Feed more reply text as it streams in. Safe after `finish` (no-op). */
append: (text: string) => void
/** Release a sealed bubble's tail without ending the turn (client-direct). */
flush?: () => void
/** No more text coming — resolves `done` once the audio drains. */
finish: () => void
/**
@@ -334,6 +336,11 @@ function openClientDirectSpeechSession(tts: DirectTtsConfig, options: VoicePlayb
ingest(false)
}
},
flush: () => {
if (!finished && !settled) {
ingest(true)
}
},
finish: () => {
if (!finished && !settled) {
finished = true