diff --git a/apps/desktop/src/lib/speech-text.test.ts b/apps/desktop/src/lib/speech-text.test.ts index 67709d7c3d..ea96cda647 100644 --- a/apps/desktop/src/lib/speech-text.test.ts +++ b/apps/desktop/src/lib/speech-text.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from 'vitest' -import { cutSentences, sanitizeTextForSpeech } from './speech-text' +import { cutSentences, IncrementalSpeechSentenceBuffer, sanitizeTextForSpeech } from './speech-text' describe('sanitizeTextForSpeech', () => { it('does not speak placeholders for fenced code blocks', () => { @@ -270,3 +270,17 @@ describe('cutSentences', () => { ]) }) }) + +describe('IncrementalSpeechSentenceBuffer', () => { + it('does not hold later sentences when prose mentions a { + const buffer = new IncrementalSpeechSentenceBuffer() + + expect(buffer.append('Wrap the plan in a tag first. ')).toEqual([ + 'Wrap the plan in a tag first.' + ]) + expect(buffer.append('Then the answer follows here in prose. And a tail')).toEqual([ + 'Then the answer follows here in prose.' + ]) + expect(buffer.flush()).toEqual(['And a tail']) + }) +}) diff --git a/apps/desktop/src/lib/speech-text.ts b/apps/desktop/src/lib/speech-text.ts index d24cf79c2f..8504586f39 100644 --- a/apps/desktop/src/lib/speech-text.ts +++ b/apps/desktop/src/lib/speech-text.ts @@ -17,7 +17,6 @@ const THINKING_PREFIX_RE = /^\s*(?:\([^)\n]{1,48}\)\s*)?(?:processing|thinking|reasoning|analyzing|pondering|contemplating|musing|cogitating|ruminating|deliberating|mulling|reflecting|computing|synthesizing|formulating|brainstorming)\.\.\.\s*/i const URL_RE = /\bhttps?:\/\/\S+/gi -const CLOSED_THINK_BLOCK_RE = /][\s\S]*?<\/think>/g const MARKDOWN_TABLE_DELIMITER_CELL_RE = /^:?-{3,}:?$/ @@ -236,28 +235,22 @@ export function cutSentences( return { sentences, rest } } -/** Incremental wrapper over cutSentences() that also hides blocks - * split across deltas. Used by the sync (non-streaming provider) fallback. */ +/** Incremental wrapper over cutSentences() for the sync (non-streaming + * provider) fallback. Deliberately a pure accumulator — like the streaming + * session's ingest (voice-playback.ts) — because the reply text it is fed is + * already text-parts-only (reasoning lives in separate parts). */ export class IncrementalSpeechSentenceBuffer { private buffer = '' - constructor(private readonly minSentenceChars?: null | number) {} - append(delta: string): string[] { - this.buffer = (this.buffer + delta).replace(CLOSED_THINK_BLOCK_RE, '') - - if (this.buffer.includes('')) { - return [] - } - - const { sentences, rest } = cutSentences(this.buffer, false, this.minSentenceChars) + const { sentences, rest } = cutSentences(this.buffer + delta, false) this.buffer = rest return sentences } flush(): string[] { - const { sentences } = cutSentences(this.buffer.replace(CLOSED_THINK_BLOCK_RE, ''), true, this.minSentenceChars) + const { sentences } = cutSentences(this.buffer, true) this.buffer = '' return sentences