From 20816c13cdde3fbc7b4b2b25ea2e3a404b487c1c Mon Sep 17 00:00:00 2001
From: Teknium <127238744+teknium1@users.noreply.github.com>
Date: Fri, 11 Sep 2026 16:35:42 -0700
Subject: [PATCH] feat(voice): pick the voice chat engine from the composer
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Switching to GPT-Live meant Settings → Voice → Voice Chat Mode, which is not
where you are when you want to talk. The composer now offers the choice where
the voice button is:
- folded layout (HUD / narrow): a "Voice chat engine" radio group in the
existing voice menu
- unfolded layout: a small chevron beside the start-voice button opening the
same rows; the button tooltip names the engine that will mount
Rows are hidden until the backend reports a mode (older gateway = no switch
that would 4002); GPT-Live is disabled with the backend's reason when no
OpenAI key resolves. Selecting writes `voice.voice_chat_mode` through
`config.set` on the LIVE gateway (local, SSH, cloud alike) and re-reads the
resolved status; it applies to the next conversation and never touches a
running one.
Backend: `voice.voice_chat_mode` joins the `config.set` word setters
(chained|gpt-live).
Live: headed desktop, chained → menu → GPT-Live → start = RTCPeerConnection
connected; menu → chained → start = no peer connection, chained controls.
---
.../src/app/chat/composer/controls.tsx | 19 +----
.../app/chat/composer/start-voice-button.tsx | 66 +++++++++++++++++
.../app/chat/composer/voice-engine-rows.tsx | 71 +++++++++++++++++++
.../src/app/chat/composer/voice-menu.tsx | 3 +
apps/desktop/src/i18n/en.ts | 7 ++
apps/desktop/src/i18n/types.ts | 7 ++
apps/desktop/src/i18n/zh.ts | 7 ++
apps/desktop/src/store/voice-live.ts | 19 +++++
.../test_config_set_voice_chat_mode.py | 38 ++++++++++
tui_gateway/methods_config_set.py | 7 +-
10 files changed, 226 insertions(+), 18 deletions(-)
create mode 100644 apps/desktop/src/app/chat/composer/start-voice-button.tsx
create mode 100644 apps/desktop/src/app/chat/composer/voice-engine-rows.tsx
create mode 100644 tests/tui_gateway/test_config_set_voice_chat_mode.py
diff --git a/apps/desktop/src/app/chat/composer/controls.tsx b/apps/desktop/src/app/chat/composer/controls.tsx
index 4538f0603a..377cbf1cda 100644
--- a/apps/desktop/src/app/chat/composer/controls.tsx
+++ b/apps/desktop/src/app/chat/composer/controls.tsx
@@ -5,7 +5,7 @@ import { Codicon } from '@/components/ui/codicon'
import { Tip, TipKeybindLabel } from '@/components/ui/tooltip'
import { useI18n } from '@/i18n'
import { triggerHaptic } from '@/lib/haptics'
-import { AudioLines, Ear, EarOff, iconSize, Layers3, Loader2, Square, Volume2, VolumeX } from '@/lib/icons'
+import { Ear, EarOff, iconSize, Layers3, Loader2, Square, Volume2, VolumeX } from '@/lib/icons'
import { cn } from '@/lib/utils'
import { $hudMode, closeHud, resetHudLayout } from '@/store/hud'
import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
@@ -13,6 +13,7 @@ import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
import { ACTIVE_ICON_BTN, GHOST_ICON_BTN, PRIMARY_ICON_BTN } from './control-classes'
import type { ConversationStatus } from './hooks/use-voice-conversation'
import { ModelPill } from './model-pill'
+import { StartVoiceButton } from './start-voice-button'
import type { ChatBarState, VoiceStatus } from './types'
import { VoiceMenu } from './voice-menu'
@@ -129,21 +130,7 @@ export function ComposerControls({
) : null}
{showVoicePrimary ? (
-
-
-
+
) : (
void }) {
+ const { t } = useI18n()
+ const engine = useVoiceEngineName()
+
+ return (
+
+
+
+
+ {engine ? (
+
+
+
+
+
+
+
+
+
+
+ ) : null}
+
+ )
+}
diff --git a/apps/desktop/src/app/chat/composer/voice-engine-rows.tsx b/apps/desktop/src/app/chat/composer/voice-engine-rows.tsx
new file mode 100644
index 0000000000..9471b8e81f
--- /dev/null
+++ b/apps/desktop/src/app/chat/composer/voice-engine-rows.tsx
@@ -0,0 +1,71 @@
+import { useStore } from '@nanostores/react'
+
+import { DropdownMenuLabel, DropdownMenuRadioGroup, DropdownMenuRadioItem, dropdownMenuRow } from '@/components/ui/dropdown-menu'
+import { useI18n } from '@/i18n'
+import { triggerHaptic } from '@/lib/haptics'
+import { notifyError } from '@/store/notifications'
+import { $voiceLiveStatus, selectedVoiceChatMode, setVoiceChatMode } from '@/store/voice-live'
+
+/**
+ * Which engine the next voice conversation mounts: the chained
+ * speech-to-text → Hermes → speech loop, or GPT-Live delegating to Hermes.
+ *
+ * Radio rows, not a toggle: the user is choosing between two named things and
+ * the checked row tells them which one the next press starts. Rendered inside
+ * whichever menu the layout has room for (the folded voice menu, or the
+ * right-click menu on the start button), so the same rows appear in both.
+ * Hidden while the backend has not answered or predates the mode, so we never
+ * offer a switch the gateway would refuse with 4002.
+ */
+export function VoiceEngineRows({ disabled }: { disabled: boolean }) {
+ const { t } = useI18n()
+ const c = t.composer
+ const status = useStore($voiceLiveStatus)
+
+ if (status === null) {
+ return null
+ }
+
+ const liveAvailable = status.available
+
+ return (
+ <>
+ {c.voiceEngine}
+ {
+ if (value !== 'chained' && value !== 'gpt-live') {
+ return
+ }
+
+ triggerHaptic('open')
+ setVoiceChatMode(value).catch(error => notifyError(error, c.voiceEngineChangeFailed))
+ }}
+ value={selectedVoiceChatMode(status)}
+ >
+
+ {c.voiceEngineChained}
+
+
+
+ {c.voiceEngineLive}
+ {liveAvailable ? null : (
+ {status.reason ?? c.voiceEngineLiveNeedsKey}
+ )}
+
+
+
+ >
+ )
+}
+
+/** Short engine name for tooltips, or null until the backend has answered. */
+export function useVoiceEngineName(): null | string {
+ const { t } = useI18n()
+ const status = useStore($voiceLiveStatus)
+
+ if (status === null) {
+ return null
+ }
+
+ return selectedVoiceChatMode(status) === 'gpt-live' ? t.composer.voiceEngineLiveShort : t.composer.voiceEngineChainedShort
+}
diff --git a/apps/desktop/src/app/chat/composer/voice-menu.tsx b/apps/desktop/src/app/chat/composer/voice-menu.tsx
index b5e371bef5..1f89f55106 100644
--- a/apps/desktop/src/app/chat/composer/voice-menu.tsx
+++ b/apps/desktop/src/app/chat/composer/voice-menu.tsx
@@ -20,6 +20,7 @@ import { $wakeWord, toggleWakeWord } from '@/store/wake-word'
import { ACTIVE_ICON_BTN, GHOST_ICON_BTN } from './control-classes'
import type { ChatBarState, VoiceStatus } from './types'
+import { VoiceEngineRows } from './voice-engine-rows'
export interface VoiceMenuProps {
autoSpeak: boolean
@@ -113,6 +114,8 @@ export function VoiceMenu({
{c.startVoice}
+
+
{/* Checkbox items, because all three are toggles the user is reading
the CURRENT state of — the reason they were pressed-state buttons
before. A plain row would fold that state away with the menu. */}
diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts
index 824430cf30..1809e97023 100644
--- a/apps/desktop/src/i18n/en.ts
+++ b/apps/desktop/src/i18n/en.ts
@@ -2796,6 +2796,13 @@ export const en: Translations = {
stopDictation: 'Stop dictation',
transcribingDictation: 'Transcribing dictation',
voiceControls: 'Voice',
+ voiceEngine: 'Voice chat engine',
+ voiceEngineChained: 'Speech-to-text + Hermes voice',
+ voiceEngineLive: 'GPT-Live (full-duplex, delegates to Hermes)',
+ voiceEngineLiveNeedsKey: 'Needs an OpenAI API key',
+ voiceEngineChangeFailed: 'Could not change the voice chat engine',
+ voiceEngineChainedShort: 'speech-to-text',
+ voiceEngineLiveShort: 'GPT-Live',
voiceDictation: 'Voice dictation',
speakReplies: 'Read replies aloud',
stopSpeakingReplies: 'Stop reading replies aloud',
diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts
index f010715dfd..9e3500ec42 100644
--- a/apps/desktop/src/i18n/types.ts
+++ b/apps/desktop/src/i18n/types.ts
@@ -2403,6 +2403,13 @@ export interface Translations {
stopDictation: string
transcribingDictation: string
voiceControls: string
+ voiceEngine: string
+ voiceEngineChained: string
+ voiceEngineLive: string
+ voiceEngineLiveNeedsKey: string
+ voiceEngineChangeFailed: string
+ voiceEngineChainedShort: string
+ voiceEngineLiveShort: string
voiceDictation: string
speakReplies: string
stopSpeakingReplies: string
diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts
index 4b85a69467..3732c42f90 100644
--- a/apps/desktop/src/i18n/zh.ts
+++ b/apps/desktop/src/i18n/zh.ts
@@ -2960,6 +2960,13 @@ export const zh: Translations = {
stopDictation: '停止听写',
transcribingDictation: '正在转写听写',
voiceControls: '语音',
+ voiceEngine: '语音聊天引擎',
+ voiceEngineChained: '语音转文字 + Hermes 语音',
+ voiceEngineLive: 'GPT-Live(全双工,委托给 Hermes)',
+ voiceEngineLiveNeedsKey: '需要 OpenAI API 密钥',
+ voiceEngineChangeFailed: '无法更改语音聊天引擎',
+ voiceEngineChainedShort: '语音转文字',
+ voiceEngineLiveShort: 'GPT-Live',
voiceDictation: '语音听写',
speakReplies: '朗读回复',
stopSpeakingReplies: '停止朗读回复',
diff --git a/apps/desktop/src/store/voice-live.ts b/apps/desktop/src/store/voice-live.ts
index fd6bbedf92..3de58e374d 100644
--- a/apps/desktop/src/store/voice-live.ts
+++ b/apps/desktop/src/store/voice-live.ts
@@ -1,6 +1,7 @@
import { atom } from 'nanostores'
import { fetchVoiceLiveStatus, type VoiceLiveStatus } from '@/lib/voice-live'
+import { activeGateway } from '@/store/gateway'
/**
* `voice.voice_chat_mode` as the backend resolves it, plus whether GPT-Live can
@@ -34,3 +35,21 @@ export async function refreshVoiceLiveStatus(): Promise
export function selectedVoiceChatMode(status: null | VoiceLiveStatus = $voiceLiveStatus.get()): 'chained' | 'gpt-live' {
return status?.mode === 'gpt-live' ? 'gpt-live' : 'chained'
}
+
+/**
+ * Persist `voice.voice_chat_mode` on the live gateway (whichever profile/host
+ * the app is talking to) and re-read the resolved status, so the menu shows
+ * what the backend will actually mount next. Takes effect on the NEXT
+ * conversation; an active one keeps its engine.
+ */
+export async function setVoiceChatMode(mode: 'chained' | 'gpt-live'): Promise {
+ const gateway = activeGateway()
+
+ if (!gateway) {
+ throw new Error('gateway not connected')
+ }
+
+ await gateway.request('config.set', { key: 'voice.voice_chat_mode', value: mode })
+
+ return refreshVoiceLiveStatus()
+}
diff --git a/tests/tui_gateway/test_config_set_voice_chat_mode.py b/tests/tui_gateway/test_config_set_voice_chat_mode.py
new file mode 100644
index 0000000000..5dee8097db
--- /dev/null
+++ b/tests/tui_gateway/test_config_set_voice_chat_mode.py
@@ -0,0 +1,38 @@
+"""`config.set voice.voice_chat_mode` is how the composer's voice menu swaps engines.
+
+The renderer's radio row writes through this key and then re-reads the resolved status; if
+the key were unlisted the handler would answer 4002 and the menu would show a switch that
+never lands on disk.
+"""
+
+import pytest
+import yaml
+
+from tui_gateway import server
+
+
+@pytest.fixture
+def config_home(tmp_path, monkeypatch):
+ monkeypatch.setattr(server, "_hermes_home", tmp_path)
+ server._cfg_cache = server._cfg_mtime = server._cfg_path = None
+ yield tmp_path / "config.yaml"
+ server._cfg_cache = server._cfg_mtime = server._cfg_path = None
+
+
+def _set(value):
+ return server._methods["config.set"](1, {"key": "voice.voice_chat_mode", "value": value})
+
+
+def test_engine_choice_reaches_the_config_file_and_round_trips(config_home):
+ assert _set("gpt-live")["result"] == {"key": "voice.voice_chat_mode", "value": "gpt-live"}
+ assert yaml.safe_load(config_home.read_text())["voice"]["voice_chat_mode"] == "gpt-live"
+
+ assert _set("Chained ")["result"]["value"] == "chained"
+ assert yaml.safe_load(config_home.read_text())["voice"]["voice_chat_mode"] == "chained"
+
+
+def test_unknown_engine_is_refused_rather_than_written(config_home):
+ answer = _set("realtime")
+
+ assert answer["error"]["code"] == 4002
+ assert not config_home.exists()
diff --git a/tui_gateway/methods_config_set.py b/tui_gateway/methods_config_set.py
index e9a6a366ae..ce5fac88c9 100644
--- a/tui_gateway/methods_config_set.py
+++ b/tui_gateway/methods_config_set.py
@@ -344,7 +344,10 @@ def _word_setters() -> dict:
lambda w: _write_config_key("display.tui_theme", w)),
# _raw_word: 0/False/[] keep their text so the error names what was sent.
"indicator": (_raw_word, INDICATOR_STYLES, "unknown indicator: {raw!r}; pick one of " + "|".join(INDICATOR_STYLES),
- lambda w: _write_config_key("display.tui_status_indicator", w))}
+ lambda w: _write_config_key("display.tui_status_indicator", w)),
+ # Which engine the desktop voice button mounts; applies to the NEXT conversation.
+ "voice.voice_chat_mode": (_word, {"chained", "gpt-live"}, "unknown voice chat mode: {value}; pick chained|gpt-live",
+ lambda w: _write_config_key("voice.voice_chat_mode", w))}
def _set_word(rid, params, key, value, session):
@@ -458,7 +461,7 @@ _CONFIG_SETTERS = {
"approval_mode": _set_approval_mode, "approvals.mode": _set_word, "yolo": _set_yolo,
"reasoning": _set_reasoning, "details_mode": _set_word, "thinking_mode": _set_word,
"density": _set_toggle, "battery": _set_toggle, "theme": _set_word,
- "statusbar": _set_toggle, "mouse": _set_toggle, "indicator": _set_word,
+ "statusbar": _set_toggle, "mouse": _set_toggle, "indicator": _set_word, "voice.voice_chat_mode": _set_word,
"cwd": _set_cwd, "terminal.cwd": _set_cwd, "workdir": _set_cwd,
"prompt": _set_prompt, "personality": _set_personality, "skin": _set_skin}