feat(voice): pick the voice chat engine from the composer

Switching to GPT-Live meant Settings → Voice → Voice Chat Mode, which is not
where you are when you want to talk. The composer now offers the choice where
the voice button is:

- folded layout (HUD / narrow): a "Voice chat engine" radio group in the
  existing voice menu
- unfolded layout: a small chevron beside the start-voice button opening the
  same rows; the button tooltip names the engine that will mount

Rows are hidden until the backend reports a mode (older gateway = no switch
that would 4002); GPT-Live is disabled with the backend's reason when no
OpenAI key resolves. Selecting writes `voice.voice_chat_mode` through
`config.set` on the LIVE gateway (local, SSH, cloud alike) and re-reads the
resolved status; it applies to the next conversation and never touches a
running one.

Backend: `voice.voice_chat_mode` joins the `config.set` word setters
(chained|gpt-live).

Live: headed desktop, chained → menu → GPT-Live → start = RTCPeerConnection
connected; menu → chained → start = no peer connection, chained controls.
This commit is contained in:
Teknium
2026-09-11 16:35:42 -07:00
parent f923faa0b8
commit 20816c13cd
10 changed files with 226 additions and 18 deletions

View File

@@ -5,7 +5,7 @@ import { Codicon } from '@/components/ui/codicon'
import { Tip, TipKeybindLabel } from '@/components/ui/tooltip'
import { useI18n } from '@/i18n'
import { triggerHaptic } from '@/lib/haptics'
import { AudioLines, Ear, EarOff, iconSize, Layers3, Loader2, Square, Volume2, VolumeX } from '@/lib/icons'
import { Ear, EarOff, iconSize, Layers3, Loader2, Square, Volume2, VolumeX } from '@/lib/icons'
import { cn } from '@/lib/utils'
import { $hudMode, closeHud, resetHudLayout } from '@/store/hud'
import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
@@ -13,6 +13,7 @@ import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
import { ACTIVE_ICON_BTN, GHOST_ICON_BTN, PRIMARY_ICON_BTN } from './control-classes'
import type { ConversationStatus } from './hooks/use-voice-conversation'
import { ModelPill } from './model-pill'
import { StartVoiceButton } from './start-voice-button'
import type { ChatBarState, VoiceStatus } from './types'
import { VoiceMenu } from './voice-menu'
@@ -129,21 +130,7 @@ export function ComposerControls({
</Tip>
) : null}
{showVoicePrimary ? (
<Tip label={c.startVoice}>
<Button
aria-label={c.startVoice}
className={PRIMARY_ICON_BTN}
disabled={disabled}
onClick={() => {
triggerHaptic('open')
conversation.onStart()
}}
size="icon"
type="button"
>
<AudioLines className={iconSize.sm} />
</Button>
</Tip>
<StartVoiceButton disabled={disabled} label={c.startVoice} onStart={conversation.onStart} />
) : (
<Tip
label={

View File

@@ -0,0 +1,66 @@
import { Button } from '@/components/ui/button'
import { DropdownMenu, DropdownMenuContent, DropdownMenuTrigger } from '@/components/ui/dropdown-menu'
import { Tip } from '@/components/ui/tooltip'
import { useI18n } from '@/i18n'
import { triggerHaptic } from '@/lib/haptics'
import { AudioLines, ChevronDown, iconSize } from '@/lib/icons'
import { cn } from '@/lib/utils'
import { GHOST_ICON_BTN, PRIMARY_ICON_BTN } from './control-classes'
import { useVoiceEngineName, VoiceEngineRows } from './voice-engine-rows'
/**
* The primary "start voice conversation" button, with the engine picker one
* click away when the layout shows the voice controls unfolded.
*
* In the folded layout the picker lives in the voice menu; unfolded there is
* no menu, so without this the only way to swap engines was Settings → Voice,
* which is not where you are when you want to talk. The chevron is a separate
* button so the primary press stays a single unambiguous action; the tooltip
* names the engine so the choice is visible before pressing.
*/
export function StartVoiceButton({ disabled, label, onStart }: { disabled: boolean; label: string; onStart: () => void }) {
const { t } = useI18n()
const engine = useVoiceEngineName()
return (
<span className="flex items-center">
<Tip label={engine ? `${label} — ${engine}` : label}>
<Button
aria-label={label}
className={cn(PRIMARY_ICON_BTN, engine && 'rounded-r-none')}
disabled={disabled}
onClick={() => {
triggerHaptic('open')
onStart()
}}
size="icon"
type="button"
>
<AudioLines className={iconSize.sm} />
</Button>
</Tip>
{engine ? (
<DropdownMenu>
<Tip label={t.composer.voiceEngine}>
<DropdownMenuTrigger asChild>
<Button
aria-label={t.composer.voiceEngine}
className={cn(GHOST_ICON_BTN, 'w-5 rounded-l-none p-0')}
disabled={disabled}
size="icon"
type="button"
variant="ghost"
>
<ChevronDown className={iconSize.xs} />
</Button>
</DropdownMenuTrigger>
</Tip>
<DropdownMenuContent align="end" className="min-w-52">
<VoiceEngineRows disabled={disabled} />
</DropdownMenuContent>
</DropdownMenu>
) : null}
</span>
)
}

View File

@@ -0,0 +1,71 @@
import { useStore } from '@nanostores/react'
import { DropdownMenuLabel, DropdownMenuRadioGroup, DropdownMenuRadioItem, dropdownMenuRow } from '@/components/ui/dropdown-menu'
import { useI18n } from '@/i18n'
import { triggerHaptic } from '@/lib/haptics'
import { notifyError } from '@/store/notifications'
import { $voiceLiveStatus, selectedVoiceChatMode, setVoiceChatMode } from '@/store/voice-live'
/**
* Which engine the next voice conversation mounts: the chained
* speech-to-text → Hermes → speech loop, or GPT-Live delegating to Hermes.
*
* Radio rows, not a toggle: the user is choosing between two named things and
* the checked row tells them which one the next press starts. Rendered inside
* whichever menu the layout has room for (the folded voice menu, or the
* right-click menu on the start button), so the same rows appear in both.
* Hidden while the backend has not answered or predates the mode, so we never
* offer a switch the gateway would refuse with 4002.
*/
export function VoiceEngineRows({ disabled }: { disabled: boolean }) {
const { t } = useI18n()
const c = t.composer
const status = useStore($voiceLiveStatus)
if (status === null) {
return null
}
const liveAvailable = status.available
return (
<>
<DropdownMenuLabel>{c.voiceEngine}</DropdownMenuLabel>
<DropdownMenuRadioGroup
onValueChange={value => {
if (value !== 'chained' && value !== 'gpt-live') {
return
}
triggerHaptic('open')
setVoiceChatMode(value).catch(error => notifyError(error, c.voiceEngineChangeFailed))
}}
value={selectedVoiceChatMode(status)}
>
<DropdownMenuRadioItem className={dropdownMenuRow} disabled={disabled} value="chained">
{c.voiceEngineChained}
</DropdownMenuRadioItem>
<DropdownMenuRadioItem className={dropdownMenuRow} disabled={disabled || !liveAvailable} value="gpt-live">
<span className="flex min-w-0 flex-col">
<span>{c.voiceEngineLive}</span>
{liveAvailable ? null : (
<span className="text-muted-foreground truncate text-xs">{status.reason ?? c.voiceEngineLiveNeedsKey}</span>
)}
</span>
</DropdownMenuRadioItem>
</DropdownMenuRadioGroup>
</>
)
}
/** Short engine name for tooltips, or null until the backend has answered. */
export function useVoiceEngineName(): null | string {
const { t } = useI18n()
const status = useStore($voiceLiveStatus)
if (status === null) {
return null
}
return selectedVoiceChatMode(status) === 'gpt-live' ? t.composer.voiceEngineLiveShort : t.composer.voiceEngineChainedShort
}

View File

@@ -20,6 +20,7 @@ import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
import { ACTIVE_ICON_BTN, GHOST_ICON_BTN } from './control-classes'
import type { ChatBarState, VoiceStatus } from './types'
import { VoiceEngineRows } from './voice-engine-rows'
export interface VoiceMenuProps {
autoSpeak: boolean
@@ -113,6 +114,8 @@ export function VoiceMenu({
{c.startVoice}
</DropdownMenuItem>
<DropdownMenuSeparator />
<VoiceEngineRows disabled={disabled} />
<DropdownMenuSeparator />
{/* Checkbox items, because all three are toggles the user is reading
the CURRENT state of — the reason they were pressed-state buttons
before. A plain row would fold that state away with the menu. */}

View File

@@ -2796,6 +2796,13 @@ export const en: Translations = {
stopDictation: 'Stop dictation',
transcribingDictation: 'Transcribing dictation',
voiceControls: 'Voice',
voiceEngine: 'Voice chat engine',
voiceEngineChained: 'Speech-to-text + Hermes voice',
voiceEngineLive: 'GPT-Live (full-duplex, delegates to Hermes)',
voiceEngineLiveNeedsKey: 'Needs an OpenAI API key',
voiceEngineChangeFailed: 'Could not change the voice chat engine',
voiceEngineChainedShort: 'speech-to-text',
voiceEngineLiveShort: 'GPT-Live',
voiceDictation: 'Voice dictation',
speakReplies: 'Read replies aloud',
stopSpeakingReplies: 'Stop reading replies aloud',

View File

@@ -2403,6 +2403,13 @@ export interface Translations {
stopDictation: string
transcribingDictation: string
voiceControls: string
voiceEngine: string
voiceEngineChained: string
voiceEngineLive: string
voiceEngineLiveNeedsKey: string
voiceEngineChangeFailed: string
voiceEngineChainedShort: string
voiceEngineLiveShort: string
voiceDictation: string
speakReplies: string
stopSpeakingReplies: string

View File

@@ -2960,6 +2960,13 @@ export const zh: Translations = {
stopDictation: '停止听写',
transcribingDictation: '正在转写听写',
voiceControls: '语音',
voiceEngine: '语音聊天引擎',
voiceEngineChained: '语音转文字 + Hermes 语音',
voiceEngineLive: 'GPT-Live(全双工,委托给 Hermes)',
voiceEngineLiveNeedsKey: '需要 OpenAI API 密钥',
voiceEngineChangeFailed: '无法更改语音聊天引擎',
voiceEngineChainedShort: '语音转文字',
voiceEngineLiveShort: 'GPT-Live',
voiceDictation: '语音听写',
speakReplies: '朗读回复',
stopSpeakingReplies: '停止朗读回复',

View File

@@ -1,6 +1,7 @@
import { atom } from 'nanostores'
import { fetchVoiceLiveStatus, type VoiceLiveStatus } from '@/lib/voice-live'
import { activeGateway } from '@/store/gateway'
/**
* `voice.voice_chat_mode` as the backend resolves it, plus whether GPT-Live can
@@ -34,3 +35,21 @@ export async function refreshVoiceLiveStatus(): Promise<null | VoiceLiveStatus>
export function selectedVoiceChatMode(status: null | VoiceLiveStatus = $voiceLiveStatus.get()): 'chained' | 'gpt-live' {
return status?.mode === 'gpt-live' ? 'gpt-live' : 'chained'
}
/**
* Persist `voice.voice_chat_mode` on the live gateway (whichever profile/host
* the app is talking to) and re-read the resolved status, so the menu shows
* what the backend will actually mount next. Takes effect on the NEXT
* conversation; an active one keeps its engine.
*/
export async function setVoiceChatMode(mode: 'chained' | 'gpt-live'): Promise<null | VoiceLiveStatus> {
const gateway = activeGateway()
if (!gateway) {
throw new Error('gateway not connected')
}
await gateway.request('config.set', { key: 'voice.voice_chat_mode', value: mode })
return refreshVoiceLiveStatus()
}

View File

@@ -0,0 +1,38 @@
"""`config.set voice.voice_chat_mode` is how the composer's voice menu swaps engines.
The renderer's radio row writes through this key and then re-reads the resolved status; if
the key were unlisted the handler would answer 4002 and the menu would show a switch that
never lands on disk.
"""
import pytest
import yaml
from tui_gateway import server
@pytest.fixture
def config_home(tmp_path, monkeypatch):
monkeypatch.setattr(server, "_hermes_home", tmp_path)
server._cfg_cache = server._cfg_mtime = server._cfg_path = None
yield tmp_path / "config.yaml"
server._cfg_cache = server._cfg_mtime = server._cfg_path = None
def _set(value):
return server._methods["config.set"](1, {"key": "voice.voice_chat_mode", "value": value})
def test_engine_choice_reaches_the_config_file_and_round_trips(config_home):
assert _set("gpt-live")["result"] == {"key": "voice.voice_chat_mode", "value": "gpt-live"}
assert yaml.safe_load(config_home.read_text())["voice"]["voice_chat_mode"] == "gpt-live"
assert _set("Chained ")["result"]["value"] == "chained"
assert yaml.safe_load(config_home.read_text())["voice"]["voice_chat_mode"] == "chained"
def test_unknown_engine_is_refused_rather_than_written(config_home):
answer = _set("realtime")
assert answer["error"]["code"] == 4002
assert not config_home.exists()

View File

@@ -344,7 +344,10 @@ def _word_setters() -> dict:
lambda w: _write_config_key("display.tui_theme", w)),
# _raw_word: 0/False/[] keep their text so the error names what was sent.
"indicator": (_raw_word, INDICATOR_STYLES, "unknown indicator: {raw!r}; pick one of " + "|".join(INDICATOR_STYLES),
lambda w: _write_config_key("display.tui_status_indicator", w))}
lambda w: _write_config_key("display.tui_status_indicator", w)),
# Which engine the desktop voice button mounts; applies to the NEXT conversation.
"voice.voice_chat_mode": (_word, {"chained", "gpt-live"}, "unknown voice chat mode: {value}; pick chained|gpt-live",
lambda w: _write_config_key("voice.voice_chat_mode", w))}
def _set_word(rid, params, key, value, session):
@@ -458,7 +461,7 @@ _CONFIG_SETTERS = {
"approval_mode": _set_approval_mode, "approvals.mode": _set_word, "yolo": _set_yolo,
"reasoning": _set_reasoning, "details_mode": _set_word, "thinking_mode": _set_word,
"density": _set_toggle, "battery": _set_toggle, "theme": _set_word,
"statusbar": _set_toggle, "mouse": _set_toggle, "indicator": _set_word,
"statusbar": _set_toggle, "mouse": _set_toggle, "indicator": _set_word, "voice.voice_chat_mode": _set_word,
"cwd": _set_cwd, "terminal.cwd": _set_cwd, "workdir": _set_cwd,
"prompt": _set_prompt, "personality": _set_personality, "skin": _set_skin}