From f923faa0b88ce5bdd0072b8cf61da1830cbb9b13 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 11 Sep 2026 03:36:27 -0700 Subject: [PATCH] =?UTF-8?q?feat(voice):=20GPT-Live=20voice=20chat=20mode?= =?UTF-8?q?=20=E2=80=94=20a=20full-duplex=20voice=20frontend=20that=20dele?= =?UTF-8?q?gates=20to=20Hermes=20(Desktop)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `voice.voice_chat_mode: gpt-live` swaps the desktop's chained STT → turn → TTS loop for OpenAI's gpt-live-1: one voice model that listens while it speaks and has no tools of its own. Every real request it hears becomes a normal Hermes turn on the open chat — any model/provider the session selected, full toolset, memory, approvals — and the voice paraphrases the reply aloud. Backend - tools/voice_live.py: mode/credential/persona resolution and the one server-side step the API needs — POST /v1/live/sessions exchanging the renderer's SDP offer, pinned to client delegation; the OpenAI key never reaches the renderer. Voice persona follows the vendor prompting guide (role, style, labelled delegation policy describing Hermes as the backend). VOICE_LIVE_TURN_NOTE is the per-turn model-input note (transcript in, speakable prose out). - REST: GET /api/audio/voice-live/status (mode + readiness, non-secret), POST /api/audio/voice-live/session (SDP exchange). The offer is passed byte-exact: a stripped trailing CRLF is a vendor 400 "unmarshal SDP: EOF". - prompt.submit accepts surface=voice-live (+ voice_context) beside hud; the note rides the model input via the existing _prepend_note seam, the persisted user row stays the user's words, the system prompt stays byte-stable. - config_defaults: voice.voice_chat_mode (chained|gpt-live), voice.gpt_live.*. Desktop - lib/voice-live.ts: RTCPeerConnection + oai-events data channel owner, transcript accumulation, session.commentary/thinking/instructions appends (500-token chunking), mute, graceful close waiting for session.closed. - hooks/use-voice-live-conversation.ts: same public shape as useVoiceConversation; delegation → prompt.submit(surface=voice-live); tool activity → quiet thinking appends; reply streamed back per sentence; spoken stop phrase ends the chat; a newer delegation interrupts an in-flight turn. - use-composer-voice mounts both engines and latches one at conversation start from the backend-resolved status; gpt-live without a key falls back to chained with a notice. Settings → Voice gets the mode dropdown, voice picker, persona. Live-verified on the worktree desktop build (headless Electron, CDP, synthetic mic): "what is 17 times 23 and which model are you on" → delegation → Hermes (Claude Sonnet 4.5 via OpenRouter) → spoken "391 … Claude Sonnet 4.5 through OpenRouter"; follow-up "double that" resolved from the spoken context → 782; "run uname -r" ran the terminal tool with "Hermes is working: terminal" fed as quiet context → spoken kernel version; "stop" closed the session (reason=close_requested). Chained mode creates no RTCPeerConnection. --- .../chat/composer/hooks/use-composer-voice.ts | 97 +++- .../hooks/use-voice-live-conversation.test.ts | 47 ++ .../hooks/use-voice-live-conversation.ts | 435 ++++++++++++++ .../app/session/hooks/use-hermes-config.ts | 3 + .../hooks/use-prompt-actions/submit.ts | 4 + .../session/hooks/use-prompt-actions/utils.ts | 7 + apps/desktop/src/app/settings/constants.ts | 26 +- apps/desktop/src/i18n/en.ts | 6 +- apps/desktop/src/i18n/types.ts | 4 + apps/desktop/src/i18n/zh.ts | 21 +- apps/desktop/src/lib/voice-live.ts | 532 ++++++++++++++++++ apps/desktop/src/store/voice-live.ts | 36 ++ hermes_cli/config_defaults.py | 13 + hermes_cli/web_models.py | 6 + hermes_cli/web_routers/audio.py | 35 +- .../tui_gateway/test_voice_live_delegation.py | 123 ++++ tools/voice_live.py | 186 ++++++ tui_gateway/methods_prompt.py | 13 +- tui_gateway/session_notifications.py | 15 +- .../docs/user-guide/features/voice-mode.md | 18 + 20 files changed, 1604 insertions(+), 23 deletions(-) create mode 100644 apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.test.ts create mode 100644 apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.ts create mode 100644 apps/desktop/src/lib/voice-live.ts create mode 100644 apps/desktop/src/store/voice-live.ts create mode 100644 tests/tui_gateway/test_voice_live_delegation.py create mode 100644 tools/voice_live.py diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts b/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts index eb8307b234..f000c7b9ab 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-voice.ts @@ -6,11 +6,13 @@ import { chatMessageText, collectUnspokenTurnSpeech } from '@/lib/chat-messages' import { triggerHaptic } from '@/lib/haptics' import { adoptSpokenReplySession, markAssistantIdSpoken, resolveSpokenReply } from '@/lib/spoken-reply' import { CONVERSATION_LEASE, READ_ALOUD_LEASE, syncTtsLease } from '@/lib/tts-lease' +import { toLiveHistory } from '@/lib/voice-live' import { clearWakeIndicator, syncWakeIndicatorWithVoice } from '@/lib/wake-indicator' import { $voiceConversationStartRequest, takeVoiceConversationStart } from '@/store/composer' import { resetBrowseState } from '@/store/composer-input-history' import { $gateway } from '@/store/gateway' import { notify, notifyError } from '@/store/notifications' +import { $voiceLiveStatus, refreshVoiceLiveStatus, selectedVoiceChatMode } from '@/store/voice-live' import { $autoSpeakReplies, $voiceStopPhrase, setAutoSpeakReplies } from '@/store/voice-prefs' import { resumeWakeAfterVoice } from '@/store/wake-word' @@ -21,6 +23,7 @@ import type { ChatBarProps } from '../types' import { useAutoSpeakReplies } from './use-auto-speak-replies' import { useVoiceConversation } from './use-voice-conversation' +import { useVoiceLiveConversation } from './use-voice-live-conversation' import { useVoiceRecorder } from './use-voice-recorder' interface UseComposerVoiceArgs { @@ -64,6 +67,9 @@ export function useComposerVoice({ // A tile's composer speaks ITS transcript, not the primary chat's. const { $messages } = useComposerScope() const [voiceConversationActive, setVoiceConversationActive] = useState(false) + // Engine selection is latched at conversation START (a Settings change + // applies to the next conversation, never mid-call). + const [liveEngineActive, setLiveEngineActive] = useState(false) const ownsWakeIndicatorRef = useRef(false) const previousSessionIdRef = useRef(sessionId) const voiceStartRequest = useStore($voiceConversationStartRequest) @@ -135,6 +141,32 @@ export function useComposerVoice({ await onSubmit(text) } + /** A GPT-Live delegation → Hermes turn. The bubble and the persisted row are + * what the user said; the transcript window rides the model input only. */ + const submitLiveDelegation = async (text: string, voiceContext: string) => { + triggerHaptic('submit') + resetBrowseState(sessionId) + clearDraft() + await onSubmit(text, { surface: 'voice-live', voiceContext }) + } + + /** Recent text turns of this chat, as GPT-Live startup history. */ + const seedLiveHistory = () => + toLiveHistory( + $messages + .get() + .filter(m => !m.hidden && (m.role === 'user' || m.role === 'assistant')) + .map(m => ({ role: m.role as 'assistant' | 'user', text: chatMessageText(m) })) + ) + + /** The tool Hermes is running right now, for quiet progress in the voice. */ + const activeToolLabel = () => { + const last = $messages.get().findLast(m => m.role === 'assistant' && !m.hidden) + const running = last?.parts.findLast(part => part.type === 'tool-call' && part.result === undefined) + + return running && running.type === 'tool-call' ? running.toolName : null + } + const wakePausedRef = useRef(false) // Resolves once the in-flight wake.pause round-trip completes (mic released by // the wake listener). The conversation awaits this before opening its own mic @@ -143,10 +175,10 @@ export function useComposerVoice({ // fail and the conversation never starts listening. const wakePauseBarrierRef = useRef | null>(null) - const conversation = useVoiceConversation({ + const chainedConversation = useVoiceConversation({ busy, consumePendingResponse, - enabled: voiceConversationActive, + enabled: voiceConversationActive && !liveEngineActive, onFatalError: () => setVoiceConversationActive(false), // Speaking over the model mid-generation interrupts the in-flight turn — // the same seam as the Stop button — so the interjection becomes the next @@ -165,6 +197,53 @@ export function useComposerVoice({ beforeMicOpen: () => wakePauseBarrierRef.current ?? undefined }) + const liveConversation = useVoiceLiveConversation({ + activeToolLabel, + beforeMicOpen: () => wakePauseBarrierRef.current ?? undefined, + busy, + consumePendingResponse, + enabled: voiceConversationActive && liveEngineActive, + onFatalError: () => setVoiceConversationActive(false), + onInterrupt, + onStopWord: () => setVoiceConversationActive(false), + onSubmit: submitLiveDelegation, + pendingResponse: pendingTurnResponse, + seedHistory: seedLiveHistory + }) + + const conversation = liveEngineActive ? liveConversation : chainedConversation + + /** Turn the conversation on with the engine `voice.voice_chat_mode` selects, + * decided in the same state batch so the other engine never sees a frame of + * `enabled`. gpt-live selected but not startable (no OpenAI key on the + * gateway) falls back to chained with a notice rather than a dead button. */ + const activateConversation = useCallback(() => { + const status = $voiceLiveStatus.get() + let live = false + + if (selectedVoiceChatMode(status) === 'gpt-live') { + if (status?.available) { + live = true + } else { + notify({ + id: 'voice-live-unavailable', + kind: 'warning', + message: t.notifications.voice.liveUnavailable(status?.reason ?? 'not configured') + }) + } + } + + setLiveEngineActive(live) + setVoiceConversationActive(true) + }, [t]) + + useEffect(() => { + if (!voiceConversationActive) { + // Prefetch so the first press picks the right engine without a round trip. + void refreshVoiceLiveStatus().catch(() => undefined) + } + }, [voiceConversationActive]) + // eslint-disable-next-line no-restricted-syntax -- ownership token used only by unmount cleanup useEffect(() => { if (target !== 'main') { @@ -197,9 +276,9 @@ export function useComposerVoice({ setVoiceConversationActive(false) void conversation.end() } else { - setVoiceConversationActive(true) + activateConversation() } - }, [conversation, disabled, voiceConversationActive]) + }, [activateConversation, conversation, disabled, voiceConversationActive]) useEffect( () => onComposerVoiceToggleRequest(toggled => toggled === target && toggleVoiceConversation()), @@ -208,9 +287,9 @@ export function useComposerVoice({ useEffect(() => { if (target === 'main' && !disabled && takeVoiceConversationStart(voiceStartRequest) && !voiceConversationActive) { - setVoiceConversationActive(true) + activateConversation() } - }, [disabled, target, voiceConversationActive, voiceStartRequest]) + }, [activateConversation, disabled, target, voiceConversationActive, voiceStartRequest]) const resumeWakeIfPaused = useCallback(() => { if (!wakePausedRef.current) { @@ -279,8 +358,8 @@ export function useComposerVoice({ // lease, and the backend unloads resident local models once no surface holds // one. Fire-and-forget — the toggle never waits on or fails from this. useEffect(() => { - void syncTtsLease(CONVERSATION_LEASE, voiceConversationActive) - }, [voiceConversationActive]) + void syncTtsLease(CONVERSATION_LEASE, voiceConversationActive && !liveEngineActive) + }, [liveEngineActive, voiceConversationActive]) useEffect(() => () => void syncTtsLease(CONVERSATION_LEASE, false), []) @@ -295,7 +374,7 @@ export function useComposerVoice({ // Explicit start/end for the on-screen conversation controls (the hotkey uses // the gated toggle above). - const startConversation = useCallback(() => setVoiceConversationActive(true), []) + const startConversation = activateConversation const endConversation = useCallback(() => { setVoiceConversationActive(false) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.test.ts b/apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.test.ts new file mode 100644 index 0000000000..5046a8f59e --- /dev/null +++ b/apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.test.ts @@ -0,0 +1,47 @@ +// @vitest-environment jsdom +import { describe, expect, it } from 'vitest' + +import { chunkForCommentary, toLiveHistory } from '@/lib/voice-live' + +import { delegationPrompt } from './use-voice-live-conversation' + +describe('GPT-Live delegation → Hermes turn', () => { + it('sends the latest user words as the turn and the exchange as model-only context', () => { + // The delegation event carries no text: both are reconstructed from + // transcript deltas, fragments of one speaker concatenated as received. + const { context, prompt } = delegationPrompt([ + { endMs: 1000, speaker: 'assistant', startMs: 0, text: 'Hi, how ' }, + { endMs: 1500, speaker: 'assistant', startMs: 1000, text: 'can I help?' }, + { endMs: 2500, speaker: 'user', startMs: 1500, text: 'What is ' }, + { endMs: 3200, speaker: 'user', startMs: 2500, text: 'the weather in Paris?' } + ]) + + expect(prompt).toBe('What is the weather in Paris?') + expect(context).toContain('Voice assistant: Hi, how can I help?') + expect(context).toContain('User: What is the weather in Paris?') + }) + + it('splits a long reply into vendor-sized commentary appends on sentence boundaries', () => { + const sentence = 'This is a sentence about the result. ' + const chunks = chunkForCommentary(sentence.repeat(80), 400) + + expect(chunks.length).toBeGreaterThan(1) + expect(chunks.every(chunk => chunk.length <= 400)).toBe(true) + expect(chunks.every(chunk => chunk.endsWith('.'))).toBe(true) + expect(chunks.join(' ')).toBe(sentence.repeat(80).trim()) + }) + + it('seeds the live session with the most recent text turns within budget', () => { + const turns = Array.from({ length: 40 }, (_, index) => ({ + role: (index % 2 === 0 ? 'user' : 'assistant') as 'assistant' | 'user', + text: `turn ${index}` + })) + + const history = toLiveHistory(turns, 6) + + expect(history).toHaveLength(6) + expect(history.at(-1)?.content[0]?.text).toBe('turn 39') + expect(history[0]?.role).toBe('user') + expect(history.find(m => m.role === 'assistant')?.content[0]?.type).toBe('output_text') + }) +}) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.ts b/apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.ts new file mode 100644 index 0000000000..d3d4a02ad4 --- /dev/null +++ b/apps/desktop/src/app/chat/composer/hooks/use-voice-live-conversation.ts @@ -0,0 +1,435 @@ +import { useCallback, useEffect, useRef, useState } from 'react' + +import { useI18n } from '@/i18n' +import { sanitizeTextForSpeech } from '@/lib/speech-text' +import { + type LiveHistoryMessage, + type LiveTranscriptFragment, + VoiceLiveSession +} from '@/lib/voice-live' +import { isVoiceStopCommand } from '@/lib/voice-stop-word' +import { notify, notifyError } from '@/store/notifications' + +import type { ConversationStatus } from './use-voice-conversation' + +/** How long an accepted delegation may sit before the gateway shows the turn running. */ +const SUBMIT_SETTLE_GRACE_MS = 15_000 +/** Quiet after the last user transcript fragment before the utterance is judged + * as a whole ("stop" ends the chat; "stop the container" is a request). */ +const UTTERANCE_SETTLE_MS = 1_500 + +interface PendingVoiceResponse { + id: string + pending: boolean + text: string +} + +interface VoiceLiveConversationOptions { + busy: boolean + enabled: boolean + onFatalError?: () => void + /** Interrupt the in-flight Hermes turn (Stop-button seam). Fired when a new + * delegation supersedes one still running. */ + onInterrupt?: () => Promise | void + onStopWord?: () => void + /** Submit a Hermes turn: `text` is the user's last words (the bubble and the + * persisted row), `voiceContext` the recent spoken exchange for the model. */ + onSubmit: (text: string, voiceContext: string) => Promise | void + pendingResponse: () => PendingVoiceResponse | null + consumePendingResponse: () => void + /** Text turns to seed the live model with when the session opens. */ + seedHistory: () => LiveHistoryMessage[] + /** Names of tools currently running in the turn (quiet progress for the voice). */ + activeToolLabel?: () => null | string + beforeMicOpen?: () => Promise | void +} + +/** Turn transcript fragments into the Hermes turn: `prompt` is what the user + * last said (the persisted user row), `context` the recent spoken exchange + * that rides the model input only (see tools/voice_live.py). */ +export function delegationPrompt(context: LiveTranscriptFragment[]): { context: string; prompt: string } { + const turns: Array<{ speaker: 'assistant' | 'user'; text: string }> = [] + + for (const fragment of context) { + const last = turns.at(-1) + + if (last && last.speaker === fragment.speaker) { + last.text += fragment.text + } else { + turns.push({ speaker: fragment.speaker, text: fragment.text }) + } + } + + const lastUser = [...turns].reverse().find(turn => turn.speaker === 'user') + const prompt = (lastUser?.text ?? '').replace(/\s+/g, ' ').trim() + + const transcript = turns + .map(turn => `${turn.speaker === 'user' ? 'User' : 'Voice assistant'}: ${turn.text.replace(/\s+/g, ' ').trim()}`) + .filter(line => !line.endsWith(': ')) + .join('\n') + + return { context: transcript, prompt: prompt || transcript.slice(-400) } +} + +/** + * GPT-Live conversation engine — same public shape as `useVoiceConversation` + * so the composer can mount either from `voice.voice_chat_mode`. + * + * Status mapping: `listening` = session up, voice idle; `speaking` = the + * remote track is producing audio; `thinking` = a delegation is in flight in + * Hermes. There is no `transcribing` phase: the voice model owns speech. + */ +export function useVoiceLiveConversation({ + busy, + enabled, + onFatalError, + onInterrupt, + onStopWord, + onSubmit, + pendingResponse, + consumePendingResponse, + seedHistory, + activeToolLabel, + beforeMicOpen +}: VoiceLiveConversationOptions) { + const { t } = useI18n() + const voiceCopy = t.notifications.voice + const [status, setStatus] = useState('idle') + const [muted, setMuted] = useState(false) + const [level, setLevel] = useState(0) + // Mirrors delegationRef for the reply-drive effect: a new delegation must + // restart the feed loop, and a ref write alone does not re-render. + const [activeDelegation, setActiveDelegation] = useState(null) + const sessionRef = useRef(null) + // Bumped by every start/end so an in-flight start() that lost the race + // (StrictMode double-effect, quick toggle) closes its session instead of + // leaving a second billed one running. + const startEpochRef = useRef(0) + const startingRef = useRef(false) + // Set at delegation submit; a turn is only "settled" once it has been seen + // running (busy) or produced a reply — the gateway ack lags the submit. + const turnObservedRef = useRef(false) + const submittedAtRef = useRef(0) + const enabledRef = useRef(enabled) + const busyRef = useRef(busy) + const speakingRef = useRef(false) + const userUtteranceRef = useRef('') + const utteranceTimerRef = useRef(null) + const delegationRef = useRef(null) + const spokenLengthRef = useRef(0) + const spokenResponseIdRef = useRef(null) + const lastToolLabelRef = useRef(null) + const wasEnabledRef = useRef(enabled) + const latest = useRef({ activeToolLabel, beforeMicOpen, onFatalError, onInterrupt, onStopWord, onSubmit, pendingResponse, consumePendingResponse, seedHistory }) + latest.current = { activeToolLabel, beforeMicOpen, onFatalError, onInterrupt, onStopWord, onSubmit, pendingResponse, consumePendingResponse, seedHistory } + + // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) + useEffect(() => { + enabledRef.current = enabled + }, [enabled]) + + // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) + useEffect(() => { + busyRef.current = busy + }, [busy]) + + const setDelegation = useCallback((id: null | string) => { + delegationRef.current = id + setActiveDelegation(id) + }, []) + + const refreshStatus = useCallback(() => { + if (!sessionRef.current) { + setStatus('idle') + + return + } + + if (speakingRef.current) { + setStatus('speaking') + } else if (delegationRef.current) { + setStatus('thinking') + } else { + setStatus('listening') + } + }, []) + + const end = useCallback(async () => { + startEpochRef.current += 1 + startingRef.current = false + + if (utteranceTimerRef.current) { + window.clearTimeout(utteranceTimerRef.current) + utteranceTimerRef.current = null + } + + userUtteranceRef.current = '' + const session = sessionRef.current + sessionRef.current = null + setDelegation(null) + spokenResponseIdRef.current = null + spokenLengthRef.current = 0 + speakingRef.current = false + session?.close() + setMuted(false) + setLevel(0) + setStatus('idle') + }, [setDelegation]) + + const start = useCallback(async () => { + if (sessionRef.current || startingRef.current) { + return + } + + startingRef.current = true + const epoch = ++startEpochRef.current + + try { + await latest.current.beforeMicOpen?.() + } catch { + // A wake-pause failure must not block an explicit start. + } + + if (!enabledRef.current || startEpochRef.current !== epoch) { + startingRef.current = false + + return + } + + const session = new VoiceLiveSession({ + // The voice model answers a bare "stop" itself (it just goes quiet) and + // never delegates it, so the spoken stop phrase is judged on the user + // transcript once the utterance settles. + onTranscript: fragment => { + if (fragment.speaker !== 'user') { + return + } + + userUtteranceRef.current += fragment.text + + if (utteranceTimerRef.current) { + window.clearTimeout(utteranceTimerRef.current) + } + + utteranceTimerRef.current = window.setTimeout(() => { + utteranceTimerRef.current = null + const utterance = userUtteranceRef.current + userUtteranceRef.current = '' + + if (sessionRef.current === session && isVoiceStopCommand(utterance)) { + void end() + latest.current.onStopWord?.() + } + }, UTTERANCE_SETTLE_MS) + }, + onClosed: (reason, usageSeconds) => { + if (sessionRef.current !== session) { + return + } + + sessionRef.current = null + setDelegation(null) + setStatus('idle') + + if (reason !== 'close_requested') { + notify({ + kind: 'warning', + message: usageSeconds != null ? `${reason} (${Math.round(usageSeconds)}s)` : reason, + title: voiceCopy.liveEnded + }) + latest.current.onFatalError?.() + } + }, + onDelegation: (delegationId, context) => { + if (sessionRef.current !== session) { + return + } + + const { context: voiceContext, prompt } = delegationPrompt(context) + + // A spoken stop command ends the conversation instead of becoming a turn. + if (prompt && isVoiceStopCommand(prompt)) { + void end() + latest.current.onStopWord?.() + + return + } + + // A newer request supersedes an in-flight turn: stop it so the answer + // the voice speaks is for what the user asked last. + if (busyRef.current) { + void latest.current.onInterrupt?.() + } + + setDelegation(delegationId) + spokenResponseIdRef.current = null + spokenLengthRef.current = 0 + lastToolLabelRef.current = null + turnObservedRef.current = false + submittedAtRef.current = Date.now() + latest.current.consumePendingResponse() + refreshStatus() + void Promise.resolve(latest.current.onSubmit(prompt, voiceContext)).catch(error => { + notifyError(error, voiceCopy.liveDelegationFailed) + session.speak(delegationId, 'Sorry, I could not reach Hermes for that request.') + setDelegation(null) + refreshStatus() + }) + }, + onError: (message, fatal) => { + notify({ kind: fatal ? 'error' : 'warning', message, title: voiceCopy.liveError }) + }, + onSpeakingChange: speaking => { + speakingRef.current = speaking + setLevel(speaking ? 0.6 : 0) + refreshStatus() + } + }) + + sessionRef.current = session + startingRef.current = false + setMuted(false) + setStatus('thinking') + + try { + await session.start(latest.current.seedHistory()) + + if (sessionRef.current !== session || startEpochRef.current !== epoch) { + session.close() + + return + } + + refreshStatus() + } catch (error) { + if (sessionRef.current === session) { + sessionRef.current = null + } + + session.close() + + if (startEpochRef.current !== epoch) { + return + } + + notifyError(error, voiceCopy.couldNotStartSession) + setStatus('idle') + latest.current.onFatalError?.() + } + }, [end, refreshStatus, setDelegation, voiceCopy.couldNotStartSession, voiceCopy.liveDelegationFailed, voiceCopy.liveEnded, voiceCopy.liveError]) + + // Drive the reply back into the voice: stream commentary as Hermes writes + // it (sentence-chunked), quiet tool progress as thinking appends, and clear + // the delegation when the turn settles. + // eslint-disable-next-line no-restricted-syntax -- turn-coordination refs (delegation id / spoken cursor), not atom mirrors + useEffect(() => { + const session = sessionRef.current + const delegationId = delegationRef.current + + if (!session || !delegationId) { + return undefined + } + + const tick = () => { + if (sessionRef.current !== session || delegationRef.current !== delegationId) { + return + } + + if (busyRef.current) { + turnObservedRef.current = true + } + + const tool = latest.current.activeToolLabel?.() ?? null + + if (tool && tool !== lastToolLabelRef.current) { + lastToolLabelRef.current = tool + session.think(delegationId, `Hermes is working: ${tool}. Not done yet.`) + } + + const response = latest.current.pendingResponse() + + if (response) { + turnObservedRef.current = true + + if (spokenResponseIdRef.current !== response.id) { + spokenResponseIdRef.current = response.id + spokenLengthRef.current = 0 + } + + const spoken = sanitizeTextForSpeech(response.text) + + // Append only completed sentences while streaming; the tail lands on settle. + if (response.pending || busyRef.current) { + const boundary = spoken.lastIndexOf('. ', spoken.length - 2) + const cut = boundary > spokenLengthRef.current ? boundary + 1 : spokenLengthRef.current + + if (cut > spokenLengthRef.current) { + session.speak(delegationId, spoken.slice(spokenLengthRef.current, cut)) + spokenLengthRef.current = cut + } + + return + } + + if (spoken.length > spokenLengthRef.current) { + session.speak(delegationId, spoken.slice(spokenLengthRef.current)) + spokenLengthRef.current = spoken.length + } + + latest.current.consumePendingResponse() + setDelegation(null) + refreshStatus() + + return + } + + // The submit ack lags: give the turn time to be seen running before + // reading "idle and no reply" as a finished turn. + if (!busyRef.current && (turnObservedRef.current || Date.now() - submittedAtRef.current > SUBMIT_SETTLE_GRACE_MS)) { + // Turn settled without a speakable reply (tool-only, error, interrupted). + if (spokenLengthRef.current === 0) { + session.think(delegationId, 'Hermes finished that request without a spoken result.') + } + + setDelegation(null) + refreshStatus() + } + } + + const timer = window.setInterval(tick, 200) + tick() + + return () => window.clearInterval(timer) + }, [activeDelegation, busy, refreshStatus, setDelegation, status]) + + const toggleMute = useCallback(() => { + setMuted(value => { + const next = !value + sessionRef.current?.setMuted(next) + + return next + }) + }, []) + + /** No explicit turn boundary in full duplex; a nudge tells the voice to answer now. */ + const stopTurn = useCallback(() => { + sessionRef.current?.instruct('The user has finished speaking. Respond now to what they said.') + }, []) + + // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) + useEffect(() => { + if (enabled && !wasEnabledRef.current) { + void start() + } + + if (!enabled && wasEnabledRef.current) { + void end() + } + + wasEnabledRef.current = enabled + }, [enabled, end, start]) + + useEffect(() => () => void end(), [end]) + + return { end, level, muted, start, status, stopTurn, toggleMute } +} diff --git a/apps/desktop/src/app/session/hooks/use-hermes-config.ts b/apps/desktop/src/app/session/hooks/use-hermes-config.ts index c560ee0d5b..ebf6a85dd9 100644 --- a/apps/desktop/src/app/session/hooks/use-hermes-config.ts +++ b/apps/desktop/src/app/session/hooks/use-hermes-config.ts @@ -16,6 +16,7 @@ import { setDefaultReasoningEffort, setIntroPersonality } from '@/store/session' +import { refreshVoiceLiveStatus } from '@/store/voice-live' import { applyAutoSpeakFromConfig, applyThinkingSoundFromConfig, @@ -147,6 +148,8 @@ export function useHermesConfig({ activeSessionIdRef }: HermesConfigOptions) { applyAutoSpeakFromConfig(config) applyVoiceStopPhraseFromConfig(config) applyThinkingSoundFromConfig(config) + // Resolved server-side (mode + whether a key resolves); non-critical. + void refreshVoiceLiveStatus().catch(() => undefined) } catch { // Config is nice-to-have; chat still works without it. } diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts index bea099da65..a47ab2e25b 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts @@ -765,6 +765,10 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { // rather than at Hermes. The gateway turns this into a per-turn hint // to read the window underneath and work in it. ...($hudMode.get() && { surface: 'hud' }), + // A GPT-Live delegation: the text is a voice transcript and the reply + // will be spoken by the voice model. Wins over HUD for this turn. + ...(options?.surface && { surface: options.surface }), + ...(options?.surface && options.voiceContext && { voice_context: options.voiceContext }), // A queue drain is a "run after" message, never a live-turn // correction. The flag tells the gateway's busy path to hold it for // the next turn untouched — without it, losing the settle race diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts index ef315bbcc7..57ca8e4293 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts @@ -719,6 +719,13 @@ export interface SubmitTextOptions { * renders anywhere — the off-screen path for widget intents. The agent * still receives the text as a normal user turn. */ displayKind?: 'hidden' + /** Per-turn client surface the gateway turns into a model-bound note. The + * HUD sets `hud` from its own store; a GPT-Live delegation passes + * `voice-live` (spoken transcript in, speakable prose out). */ + surface?: 'voice-live' + /** With `surface: 'voice-live'`: the recent spoken exchange, appended to the + * model-bound note by the gateway (never persisted, never rendered). */ + voiceContext?: string fromQueue?: boolean /** Runtime session id to submit into. Queue drains pass this so a * backgrounded/source session cannot be replaced by the current foreground diff --git a/apps/desktop/src/app/settings/constants.ts b/apps/desktop/src/app/settings/constants.ts index daa1d05cc3..3a344aca97 100644 --- a/apps/desktop/src/app/settings/constants.ts +++ b/apps/desktop/src/app/settings/constants.ts @@ -251,6 +251,13 @@ export const ENUM_OPTIONS: Record = { // Speech-to-text backends — kept in sync with the stt block in // hermes_cli/config.py (local/groq/openai/mistral/elevenlabs). 'stt.provider': ['local', 'groq', 'openai', 'mistral', 'xai', 'elevenlabs'], + // How the desktop voice conversation is wired — tools/voice_live.py owns the + // gpt-live branch (one full-duplex voice model delegating to Hermes). + 'voice.voice_chat_mode': ['chained', 'gpt-live'], + 'voice.gpt_live.voice': [ + 'marin', 'cedar', 'quartz', 'ripple', 'vesper', 'willow', 'stone', 'gleam', 'meridian', + 'bossa', 'tempo', 'beacon', 'delta', 'cinder' + ], // OpenAI TTS voices — the union across models (per the OpenAI TTS API // docs). Model-specific narrowing happens in enumOptionsFor(): // tts-1 / tts-1-hd support 9 voices; gpt-4o-mini-tts supports all 13. @@ -355,6 +362,7 @@ export const ENUM_OPTIONS: Record = { // suggestions rather than a gate for these keys. export const FREE_INPUT_KEYS = new Set([ 'tts.edge.voice', + 'voice.gpt_live.voice', 'tts.openai.model', 'tts.openai.voice', 'tts.elevenlabs.voice_id', @@ -437,7 +445,12 @@ export const FIELD_LABELS: Record = defineFieldCopy({ voice: { recordKey: 'Voice Shortcut', maxRecordingSeconds: 'Max Recording Length', - autoTts: 'Read Responses Aloud' + autoTts: 'Read Responses Aloud', + voiceChatMode: 'Voice Chat Mode', + gptLive: { + voice: 'GPT-Live Voice', + instructions: 'GPT-Live Persona' + } }, stt: { enabled: 'Speech To Text', @@ -598,7 +611,13 @@ export const FIELD_DESCRIPTIONS: Record = defineFieldCopy({ enabled: 'Summarize older context when conversations get large.' }, voice: { - autoTts: 'Automatically speak assistant responses.' + autoTts: 'Automatically speak assistant responses.', + voiceChatMode: + 'chained: speech-to-text → Hermes → text-to-speech with the providers below. gpt-live: one full-duplex OpenAI voice model (gpt-live-1) listens and talks, and hands every real request to Hermes — any model you have selected answers with the full toolset. Needs an OpenAI API key; the voice layer bills $0.05 per minute.', + gptLive: { + voice: 'Voice for GPT-Live mode. Custom voice IDs are accepted.', + instructions: 'Extra sentences for the live voice persona (tone, pace, language). Hermes keeps its own system prompt.' + } }, tts: { xai: { @@ -704,6 +723,9 @@ export const SECTIONS: DesktopConfigSection[] = [ label: 'Voice', icon: Mic, keys: [ + 'voice.voice_chat_mode', + 'voice.gpt_live.voice', + 'voice.gpt_live.instructions', 'tts.provider', 'stt.enabled', 'stt.echo_transcripts', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 304627c190..824430cf30 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -246,7 +246,11 @@ export const en: Translations = { transcriptionFailed: 'Voice transcription failed', transcriptionUnavailable: 'Voice transcription is not available yet.', tryRecordingAgain: 'Try recording again.', - unavailable: 'Voice unavailable' + unavailable: 'Voice unavailable', + liveEnded: 'Live voice session ended', + liveError: 'Live voice', + liveDelegationFailed: 'Could not hand the request to Hermes', + liveUnavailable: reason => `GPT-Live voice chat is not available: ${reason}. Using speech-to-text instead.` }, native: { approvalTitle: 'Approval needed', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 861a429854..f010715dfd 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -285,6 +285,10 @@ export interface Translations { transcriptionUnavailable: string tryRecordingAgain: string unavailable: string + liveEnded: string + liveError: string + liveDelegationFailed: string + liveUnavailable: (reason: string) => string } // Native OS notification copy (titles + generic fallback bodies). Dynamic // bodies (the agent's reply, a command, an error) are passed through raw. diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 895cb83c3b..4b85a69467 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -240,7 +240,11 @@ export const zh: Translations = { transcriptionFailed: '语音转写失败', transcriptionUnavailable: '语音转写暂不可用。', tryRecordingAgain: '请再录一次。', - unavailable: '语音不可用' + unavailable: '语音不可用', + liveEnded: '实时语音会话已结束', + liveError: '实时语音', + liveDelegationFailed: '无法将请求交给 Hermes', + liveUnavailable: reason => `GPT-Live 语音聊天不可用:${reason}。已改用语音转文字。` }, native: { approvalTitle: '需要批准', @@ -859,7 +863,12 @@ export const zh: Translations = { voice: { recordKey: '语音快捷键', maxRecordingSeconds: '最长录音时长', - autoTts: '朗读回复' + autoTts: '朗读回复', + voiceChatMode: '语音聊天模式', + gptLive: { + voice: 'GPT-Live 音色', + instructions: 'GPT-Live 人设' + } }, stt: { enabled: '语音转文字', @@ -1006,7 +1015,13 @@ export const zh: Translations = { enabled: '当对话变大时对较早的上下文进行摘要。' }, voice: { - autoTts: '自动朗读助手回复。' + autoTts: '自动朗读助手回复。', + voiceChatMode: + 'chained:语音转文字 → Hermes → 文字转语音,使用下方的提供商。gpt-live:一个全双工的 OpenAI 语音模型(gpt-live-1)负责听和说,并把每个实际请求交给 Hermes——由你选择的任意模型带着完整工具集作答。需要 OpenAI API 密钥;语音层按每分钟 $0.05 计费。', + gptLive: { + voice: 'GPT-Live 模式使用的音色,可填写自定义音色 ID。', + instructions: '附加到实时语音人设的句子(语气、语速、语言)。Hermes 保留自己的系统提示词。' + } }, stt: { enabled: '启用本地或提供方支持的语音转写。', diff --git a/apps/desktop/src/lib/voice-live.ts b/apps/desktop/src/lib/voice-live.ts new file mode 100644 index 0000000000..aca8e4aae3 --- /dev/null +++ b/apps/desktop/src/lib/voice-live.ts @@ -0,0 +1,532 @@ +import { profileScoped } from '@/api/client' +import { hermesApi } from '@/hermes' + +/** + * GPT-Live voice chat: the full-duplex voice frontend that DELEGATES to Hermes. + * + * `voice.voice_chat_mode: gpt-live` swaps the chained mic → STT → turn → TTS + * loop for one OpenAI voice model (`gpt-live-1`) that listens and speaks at + * the same time over WebRTC and has no tools of its own. Whenever the user + * asks for real work it emits `session.delegation.created`; the desktop turns + * that into an ordinary Hermes turn on the open session and streams the reply + * back with `session.commentary.append`, which the voice paraphrases aloud. + * Hermes keeps every capability — model choice, tools, memory, approvals. + * + * This module owns the transport only: session creation via the gateway + * (the OpenAI key never reaches the renderer), the RTCPeerConnection, the + * `oai-events` data channel, transcript accumulation and the command + * surface the conversation hook drives. Vendor contract: + * https://developers.openai.com/api/docs/guides/live-delegation + */ + +export type VoiceChatMode = 'chained' | 'gpt-live' + +export interface VoiceLiveStatus { + mode: VoiceChatMode + available: boolean + reason: null | string + model: string + voice: string +} + +export interface LiveHistoryMessage { + type: 'message' + role: 'assistant' | 'developer' | 'user' + content: Array<{ type: 'input_text' | 'output_text'; text: string }> +} + +interface LiveServerEvent { + type: string + event_id?: string + client_event_id?: string + delta?: string + start_ms?: number + end_ms?: number + delegation?: { id: string; type: string; target: string } + error?: { type?: string; code?: null | string; message?: string; client_event_id?: string } + usage?: { seconds?: number } + reason?: string + session?: { id: string } +} + +export interface LiveTranscriptFragment { + speaker: 'assistant' | 'user' + text: string + startMs: number + endMs: number +} + +export interface VoiceLiveHandlers { + /** GPT-Live asked the backend (Hermes) for help. `context` is the recent + * transcript window, newest last — the delegation itself carries no text. */ + onDelegation: (delegationId: string, context: LiveTranscriptFragment[]) => void + /** Vendor-side error. `fatal` when the session is gone. */ + onError: (message: string, fatal: boolean) => void + /** `session.closed` arrived (or the transport dropped without it). */ + onClosed: (reason: string, usageSeconds: null | number) => void + /** Transcript deltas, for captions / live UI. */ + onTranscript?: (fragment: LiveTranscriptFragment) => void + /** Assistant audio output level hint: the remote track is speaking. */ + onSpeakingChange?: (speaking: boolean) => void +} + +const CLOSE_TIMEOUT_MS = 15_000 +const ICE_GATHER_TIMEOUT_MS = 10_000 +// Vendor cap: 500 tokens per append. ~4 chars/token, keep headroom. +const APPEND_CHAR_LIMIT = 1_400 +// How much conversation the backend receives per delegation. +const CONTEXT_WINDOW_MS = 5 * 60_000 +const CONTEXT_MAX_FRAGMENTS = 80 + +export async function fetchVoiceLiveStatus(): Promise { + try { + const response = await hermesApi<{ ok: boolean } & VoiceLiveStatus>({ + ...profileScoped(), + path: '/api/audio/voice-live/status' + }) + + if (!response?.ok) { + return null + } + + return { + available: Boolean(response.available), + mode: response.mode === 'gpt-live' ? 'gpt-live' : 'chained', + model: response.model, + reason: response.reason ?? null, + voice: response.voice + } + } catch { + // Older backend without the endpoint → chained. + return null + } +} + +/** Split a reply into append-sized chunks on sentence boundaries. */ +export function chunkForCommentary(text: string, limit = APPEND_CHAR_LIMIT): string[] { + const clean = text.replace(/\s+/g, ' ').trim() + + if (!clean) { + return [] + } + + if (clean.length <= limit) { + return [clean] + } + + const chunks: string[] = [] + let current = '' + + for (const sentence of clean.split(/(?<=[.!?])\s+/)) { + if (sentence.length > limit) { + if (current) { + chunks.push(current) + current = '' + } + + for (let index = 0; index < sentence.length; index += limit) { + chunks.push(sentence.slice(index, index + limit)) + } + + continue + } + + const candidate = current ? `${current} ${sentence}` : sentence + + if (candidate.length > limit) { + chunks.push(current) + current = sentence + } else { + current = candidate + } + } + + if (current) { + chunks.push(current) + } + + return chunks +} + +/** Seed history for a new Live session from the chat transcript (text turns only). */ +export function toLiveHistory( + turns: Array<{ role: 'assistant' | 'user'; text: string }>, + maxMessages = 24, + maxChars = 6_000 +): LiveHistoryMessage[] { + const out: LiveHistoryMessage[] = [] + let budget = maxChars + + for (const turn of [...turns].reverse()) { + const text = turn.text.replace(/\s+/g, ' ').trim().slice(0, 1_200) + + if (!text) { + continue + } + + if (out.length >= maxMessages || budget - text.length < 0) { + break + } + + budget -= text.length + out.unshift({ + content: [{ text, type: turn.role === 'assistant' ? 'output_text' : 'input_text' }], + role: turn.role, + type: 'message' + }) + } + + return out +} + +async function waitForIceGathering(connection: RTCPeerConnection): Promise { + if (connection.iceGatheringState === 'complete') { + return + } + + await new Promise((resolve, reject) => { + const timeout = window.setTimeout(() => { + connection.removeEventListener('icegatheringstatechange', onState) + // Trickle is fine: the vendor answers with the candidates it has. + resolve() + }, ICE_GATHER_TIMEOUT_MS) + + function onState() { + if (connection.iceGatheringState !== 'complete') { + return + } + + window.clearTimeout(timeout) + connection.removeEventListener('icegatheringstatechange', onState) + resolve() + } + + connection.addEventListener('icegatheringstatechange', onState) + connection.addEventListener('connectionstatechange', () => { + if (connection.connectionState === 'failed') { + window.clearTimeout(timeout) + reject(new Error('WebRTC connection failed')) + } + }) + }) +} + +export class VoiceLiveSession { + readonly audio: HTMLAudioElement + private peer: null | RTCPeerConnection = null + private events: null | RTCDataChannel = null + private microphone: null | MediaStream = null + private closeTimer: null | number = null + private finalized = false + private started = false + private eventCounter = 0 + private transcript: LiveTranscriptFragment[] = [] + private speakingProbe: null | number = null + private analyser: null | AnalyserNode = null + private audioContext: null | AudioContext = null + private lastSpeaking = false + sessionId: null | string = null + /** The delegation currently being answered by Hermes; late results for an + * older id are dropped by the conversation hook. */ + activeDelegationId: null | string = null + + constructor(private readonly handlers: VoiceLiveHandlers) { + this.audio = new Audio() + this.audio.autoplay = true + } + + get connected(): boolean { + return this.started && this.events?.readyState === 'open' + } + + private nextEventId(prefix: string): string { + this.eventCounter += 1 + + return `${prefix}_${this.eventCounter}` + } + + private send(event: Record): boolean { + if (!this.events || this.events.readyState !== 'open') { + return false + } + + this.events.send(JSON.stringify(event)) + + return true + } + + /** Recent conversation, oldest first, bounded by time and count. */ + contextWindow(): LiveTranscriptFragment[] { + const last = this.transcript.at(-1) + + if (!last) { + return [] + } + + const floor = last.endMs - CONTEXT_WINDOW_MS + + return this.transcript.filter(fragment => fragment.endMs >= floor).slice(-CONTEXT_MAX_FRAGMENTS) + } + + async start(history: LiveHistoryMessage[]): Promise { + if (this.peer) { + throw new Error('GPT-Live session already started') + } + + const connection = new RTCPeerConnection() + this.peer = connection + + connection.addEventListener('track', event => { + const stream = new MediaStream([event.track]) + this.audio.srcObject = stream + void this.audio.play().catch(() => undefined) + this.armSpeakingProbe(stream) + }) + connection.addEventListener('connectionstatechange', () => { + if (connection.connectionState === 'failed' || connection.connectionState === 'disconnected') { + this.finish('connection_lost', null) + } + }) + + this.microphone = await navigator.mediaDevices.getUserMedia({ + audio: { autoGainControl: true, echoCancellation: true, noiseSuppression: true } + }) + + for (const track of this.microphone.getAudioTracks()) { + connection.addTrack(track, this.microphone) + } + + // Register the data channel before the offer so its m-line is negotiated. + const events = connection.createDataChannel('oai-events') + this.events = events + events.addEventListener('message', ({ data }) => this.handleEvent(String(data))) + events.addEventListener('close', () => { + if (!this.finalized) { + this.finish('connection_lost', null) + } + }) + + const offer = await connection.createOffer() + await connection.setLocalDescription(offer) + await waitForIceGathering(connection) + + const sdp = connection.localDescription?.sdp + + if (!sdp) { + throw new Error('Missing local SDP offer') + } + + const response = await hermesApi<{ + ok: boolean + session?: { id: string } + transport?: { sdp: string; type: string } + }>({ + ...profileScoped(), + body: { history, sdp }, + method: 'POST', + path: '/api/audio/voice-live/session', + timeoutMs: 45_000 + }) + + if (!response?.ok || !response.transport?.sdp) { + throw new Error('GPT-Live session creation failed') + } + + this.sessionId = response.session?.id ?? null + await connection.setRemoteDescription({ sdp: response.transport.sdp, type: 'answer' }) + } + + private armSpeakingProbe(stream: MediaStream): void { + try { + const context = new AudioContext() + const source = context.createMediaStreamSource(stream) + const analyser = context.createAnalyser() + analyser.fftSize = 512 + source.connect(analyser) + this.audioContext = context + this.analyser = analyser + const buffer = new Uint8Array(analyser.frequencyBinCount) + let quietFrames = 0 + + this.speakingProbe = window.setInterval(() => { + analyser.getByteTimeDomainData(buffer) + let peak = 0 + + for (const sample of buffer) { + peak = Math.max(peak, Math.abs(sample - 128)) + } + + const loud = peak > 6 + quietFrames = loud ? 0 : quietFrames + 1 + const speaking = loud || quietFrames < 4 + + if (speaking !== this.lastSpeaking) { + this.lastSpeaking = speaking + this.handlers.onSpeakingChange?.(speaking) + } + }, 100) + } catch { + // No analyser → no speaking indicator; the conversation still works. + } + } + + private handleEvent(raw: string): void { + let event: LiveServerEvent + + try { + event = JSON.parse(raw) as LiveServerEvent + } catch { + return + } + + switch (event.type) { + case 'session.started': + this.started = true + this.sessionId = event.session?.id ?? this.sessionId + + return + + case 'session.input_transcript.delta': + case 'session.output_transcript.delta': { + const fragment: LiveTranscriptFragment = { + endMs: event.end_ms ?? 0, + speaker: event.type === 'session.input_transcript.delta' ? 'user' : 'assistant', + startMs: event.start_ms ?? 0, + text: event.delta ?? '' + } + + this.transcript.push(fragment) + + if (this.transcript.length > 2_000) { + this.transcript.splice(0, this.transcript.length - 1_500) + } + + this.handlers.onTranscript?.(fragment) + + return + } + + case 'session.delegation.created': { + const id = event.delegation?.id + + if (id) { + this.activeDelegationId = id + this.handlers.onDelegation(id, this.contextWindow()) + } + + return + } + + case 'error': { + const code = event.error?.code ?? '' + + // Late appends after our own close are expected noise. + if (code === 'context_injection_incomplete') { + return + } + + this.handlers.onError(event.error?.message ?? 'GPT-Live error', false) + + return + } + + case 'session.closed': + this.finish(event.reason ?? 'closed', event.usage?.seconds ?? null) + + return + + default: + return + } + } + + /** Quiet progress for the live model ("Hermes is running the tests…"). */ + think(delegationId: null | string, content: string): void { + const text = content.replace(/\s+/g, ' ').trim().slice(0, APPEND_CHAR_LIMIT) + + if (text) { + this.send({ + content: text, + delegation_id: delegationId, + event_id: this.nextEventId('think'), + type: 'session.thinking.append' + }) + } + } + + /** A result the voice should say aloud (paraphrased). */ + speak(delegationId: null | string, content: string): void { + for (const chunk of chunkForCommentary(content)) { + this.send({ + content: chunk, + delegation_id: delegationId, + event_id: this.nextEventId('say'), + type: 'session.commentary.append' + }) + } + } + + /** Steer the live persona mid-conversation (session-wide). */ + instruct(content: string): void { + const text = content.trim().slice(0, APPEND_CHAR_LIMIT) + + if (text) { + this.send({ + content: text, + delegation_id: null, + event_id: this.nextEventId('instr'), + type: 'session.instructions.append' + }) + } + } + + setMuted(muted: boolean): void { + for (const track of this.microphone?.getAudioTracks() ?? []) { + track.enabled = !muted + } + + this.send({ event_id: this.nextEventId(muted ? 'mute' : 'unmute'), type: muted ? 'session.input_audio.mute' : 'session.input_audio.unmute' }) + } + + /** Graceful close: ask for `session.closed`, tear down after it (or a timeout). */ + close(): void { + if (this.finalized) { + return + } + + if (!this.send({ type: 'session.close' })) { + this.finish('close_requested', null) + + return + } + + this.closeTimer = window.setTimeout(() => this.finish('close_requested', null), CLOSE_TIMEOUT_MS) + } + + private finish(reason: string, usageSeconds: null | number): void { + if (this.finalized) { + return + } + + this.finalized = true + + if (this.closeTimer) { + window.clearTimeout(this.closeTimer) + this.closeTimer = null + } + + if (this.speakingProbe) { + window.clearInterval(this.speakingProbe) + this.speakingProbe = null + } + + this.analyser?.disconnect() + void this.audioContext?.close().catch(() => undefined) + this.microphone?.getTracks().forEach(track => track.stop()) + this.events?.close() + this.peer?.close() + this.audio.srcObject = null + this.audio.pause() + this.handlers.onClosed(reason, usageSeconds) + } +} diff --git a/apps/desktop/src/store/voice-live.ts b/apps/desktop/src/store/voice-live.ts new file mode 100644 index 0000000000..fd6bbedf92 --- /dev/null +++ b/apps/desktop/src/store/voice-live.ts @@ -0,0 +1,36 @@ +import { atom } from 'nanostores' + +import { fetchVoiceLiveStatus, type VoiceLiveStatus } from '@/lib/voice-live' + +/** + * `voice.voice_chat_mode` as the backend resolves it, plus whether GPT-Live can + * actually start (an OpenAI key resolves on the gateway host). The composer + * mounts the chained or the live conversation engine from this; refreshed with + * the config snapshot so a Settings change applies to the next conversation. + */ +export const $voiceLiveStatus = atom(null) + +let inflight: null | Promise = null + +export async function refreshVoiceLiveStatus(): Promise { + if (inflight) { + return inflight + } + + inflight = fetchVoiceLiveStatus() + .then(status => { + $voiceLiveStatus.set(status) + + return status + }) + .finally(() => { + inflight = null + }) + + return inflight +} + +/** Selected mode. `chained` until the backend answers, or when the backend predates the mode. */ +export function selectedVoiceChatMode(status: null | VoiceLiveStatus = $voiceLiveStatus.get()): 'chained' | 'gpt-live' { + return status?.mode === 'gpt-live' ? 'gpt-live' : 'chained' +} diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 6111e98eb2..4a6fbb0fd4 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1116,6 +1116,19 @@ DEFAULT_CONFIG = { }, "voice": { + # How the Desktop voice conversation is wired: + # chained — STT → Hermes turn → TTS (the stt.* / tts.* providers below) + # gpt-live — one full-duplex voice model (OpenAI GPT-Live) owns the mic and speaker and + # DELEGATES every real request to Hermes (any model / provider you have + # selected); needs an OpenAI API key. $0.05/min voice layer billing. + "voice_chat_mode": "chained", + "gpt_live": { + "model": "gpt-live-1", + "voice": "marin", # marin | quartz | ripple | vesper | willow | stone | gleam | meridian | ... + # Extra sentences appended to the live model's conversation persona (tone, pacing, language). + "instructions": "", + # optional "api_key" / "base_url" keys override the OpenAI audio credentials for this mode only + }, "record_key": "ctrl+b", "submit_mode": "direct", # TUI: direct submits immediately; draft = editable transcript "max_recording_seconds": 120, diff --git a/hermes_cli/web_models.py b/hermes_cli/web_models.py index 3f4faf5ebd..8aeacb378b 100644 --- a/hermes_cli/web_models.py +++ b/hermes_cli/web_models.py @@ -217,6 +217,12 @@ class DebugShareRequest(BaseModel): class TTSSpeakRequest(BaseModel): text: str +class VoiceLiveSessionRequest(BaseModel): + """POST /api/audio/voice-live/session: the renderer's WebRTC SDP offer plus optional prior + text turns (``{"type":"message","role":..,"content":[..]}``) to seed the live voice model.""" + sdp: str + history: Optional[List[Dict[str, Any]]] = None + class TTSLeaseRequest(BaseModel): """POST /api/audio/tts-lease: ``lease`` names the toggle/surface holding the lease (``desktop:read-aloud``, ``desktop:conversation``); ``active`` True acquires + warms, False releases.""" diff --git a/hermes_cli/web_routers/audio.py b/hermes_cli/web_routers/audio.py index ec49c90467..d70b194bbc 100644 --- a/hermes_cli/web_routers/audio.py +++ b/hermes_cli/web_routers/audio.py @@ -22,7 +22,7 @@ from hermes_cli.web_deps import late from hermes_cli.web_server_chat import _ws_auth_ok, _ws_request_is_allowed from hermes_cli.web_server_gateway import _split_text_for_speak_stream from fastapi import HTTPException, WebSocket, WebSocketDisconnect -from hermes_cli.web_models import AudioTranscriptionRequest, TTSSpeakRequest, TTSLeaseRequest +from hermes_cli.web_models import AudioTranscriptionRequest, TTSSpeakRequest, TTSLeaseRequest, VoiceLiveSessionRequest from typing import Any, Dict, Optional _log = logging.getLogger("hermes_cli.web_server") @@ -164,6 +164,39 @@ async def get_client_voice_config(profile: Optional[str] = None): return {"ok": True, **result} +@router.get("/api/audio/voice-live/status") +async def get_voice_live_status(profile: Optional[str] = None): + """Which voice chat mode the profile selected (``chained`` | ``gpt-live``) and whether GPT-Live + can start. Non-secret: the desktop decides which conversation engine to mount from this.""" + from tools.voice_live import resolve_gpt_live_status + with http_failure("GPT-Live status resolution failed", 500, "GPT-Live status failed"): + result = await _run_config_scoped(profile, resolve_gpt_live_status) + return {"ok": True, **result} + + +@router.post("/api/audio/voice-live/session") +async def create_voice_live_session(payload: VoiceLiveSessionRequest, profile: Optional[str] = None): + """Exchange the renderer's WebRTC SDP offer for a GPT-Live session answer. + + The project API key stays on this host; the renderer only receives the session id and the + SDP answer. Client delegation is fixed at creation: every ``session.delegation.created`` the + renderer receives becomes a Hermes turn on the session it belongs to. + """ + from tools.voice_live import create_webrtc_session + # Validate emptiness only: the vendor's SDP parser needs the offer byte-exact, including the + # trailing CRLF (a stripped offer answers 400 "failed to unmarshal SDP: EOF"). + sdp = payload.sdp or "" + if not sdp.strip(): + raise HTTPException(status_code=400, detail="An SDP offer is required") + try: + result = await _run_config_scoped(profile, lambda: create_webrtc_session(sdp, payload.history)) + except ValueError as exc: + raise HTTPException(status_code=503, detail=str(exc)) + except RuntimeError as exc: + raise HTTPException(status_code=502, detail=str(exc)) + return {"ok": True, **result} + + def _elevenlabs_voice_label(voice: Dict[str, Any]) -> str: name = str(voice.get("name") or voice.get("voice_id") or "Voice").strip() category = str(voice.get("category") or "").strip() diff --git a/tests/tui_gateway/test_voice_live_delegation.py b/tests/tui_gateway/test_voice_live_delegation.py new file mode 100644 index 0000000000..34ba98f14b --- /dev/null +++ b/tests/tui_gateway/test_voice_live_delegation.py @@ -0,0 +1,123 @@ +"""GPT-Live voice chat mode: the full-duplex voice frontend that delegates to Hermes. + +The live voice model owns the microphone and speaker and has no tools; every real +request is delegated to Hermes as a normal turn on the open session. Two contracts +matter and are pinned here: + +* the gateway never hands the OpenAI key to the renderer — ``POST /v1/live/sessions`` + is performed server-side from the renderer's SDP offer, with the session pinned to + client delegation so Hermes (any model) is the backend; +* a turn submitted from the live voice surface carries the spoken-delegation note on + the MODEL INPUT only (the byte-stable system prompt is untouched), exactly like the + HUD note it sits beside. +""" + +import json +import threading +import types + +import pytest + +from tools import voice_live +from tui_gateway import server + + +def _session(**extra): + return { + "agent": types.SimpleNamespace(valid_tool_names=set()), + "session_key": "session-key", + "history": [], + "history_lock": threading.Lock(), + "history_version": 0, + "running": True, + "transport": None, + "attached_images": [], + **extra, + } + + +class TestSessionCreation: + def test_client_delegation_and_key_stay_server_side(self, monkeypatch): + """Whatever the renderer sends, the vendor request pins ``delegation.type == client`` + (Hermes is the backend) and authenticates with the resolved key; the client only ever + sees the vendor answer.""" + captured = {} + + class _Resp: + def __init__(self, body): + self._body = body + + def read(self): + return self._body + + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def fake_urlopen(req, timeout=0): + captured["url"] = req.full_url + captured["auth"] = req.get_header("Authorization") + captured["body"] = json.loads(req.data) + return _Resp(json.dumps({"session": {"id": "live_x"}, "transport": {"type": "webrtc", "sdp": "answer"}}).encode()) + + monkeypatch.setattr(voice_live.urllib.request, "urlopen", fake_urlopen) + monkeypatch.setattr(voice_live, "_live_section", lambda voice=None: {"voice": "willow", "instructions": "Speak Spanish."}) + monkeypatch.setattr(voice_live, "_resolve_credentials", lambda live: ("sk-test", "https://api.example/v1")) + + result = voice_live.create_webrtc_session("v=0 offer", history=[{"type": "message", "role": "user", "content": []}]) + + assert result["transport"]["sdp"] == "answer" + assert captured["url"] == "https://api.example/v1/live/sessions" + assert captured["auth"] == "Bearer sk-test" + session = captured["body"]["session"] + assert session["delegation"] == {"type": "client"} + assert session["audio"]["output"]["voice"] == "willow" + assert session["instructions"].endswith("Speak Spanish.") + assert session["input"][0]["role"] == "user" + assert captured["body"]["transport"] == {"type": "webrtc", "sdp": "v=0 offer"} + assert "sk-test" not in json.dumps(result) + + def test_missing_key_refuses_before_any_network(self, monkeypatch): + monkeypatch.setattr(voice_live, "_resolve_credentials", lambda live: ("", voice_live.DEFAULT_LIVE_BASE_URL)) + monkeypatch.setattr(voice_live.urllib.request, "urlopen", lambda *a, **k: pytest.fail("must not call the vendor")) + + with pytest.raises(ValueError): + voice_live.create_webrtc_session("v=0 offer") + assert voice_live.resolve_gpt_live_status()["available"] is False + + +class TestVoiceLiveTurnNote: + @pytest.fixture + def busy_session(self): + session = _session() + server._sessions["sid"] = session + yield session + server._sessions.pop("sid", None) + + def test_live_surface_recorded_and_noted_with_spoken_context(self, busy_session): + """The persisted row is the user's words; the transcript window reaches the model only.""" + server._methods["prompt.submit"]( + "r1", {"session_id": "sid", "text": "what's the weather", "queued": True, "surface": "voice-live", + "voice_context": "Voice assistant: Hi\nUser: what's the weather"}) + + assert busy_session["client_surface"] == "voice-live" + note = server._hud_surface_note(busy_session) + assert note.startswith(voice_live.VOICE_LIVE_TURN_NOTE) + assert "spoken" in note and "no markdown" in note + assert "User: what's the weather" in note + + def test_voice_context_ignored_off_the_live_surface(self, busy_session): + server._methods["prompt.submit"]( + "r1", {"session_id": "sid", "text": "x", "queued": True, "voice_context": "User: smuggled"}) + + assert busy_session["voice_live_context"] == "" + assert server._hud_surface_note(busy_session) == "" + + def test_plain_window_submit_clears_the_live_surface(self, busy_session): + server._methods["prompt.submit"]("r1", {"session_id": "sid", "text": "x", "queued": True, "surface": "voice-live"}) + server._methods["prompt.submit"]("r2", {"session_id": "sid", "text": "y", "queued": True}) + + assert busy_session["client_surface"] == "" + assert server._hud_surface_note(busy_session) == "" diff --git a/tools/voice_live.py b/tools/voice_live.py new file mode 100644 index 0000000000..b9ab6f7863 --- /dev/null +++ b/tools/voice_live.py @@ -0,0 +1,186 @@ +"""GPT-Live voice chat mode: the full-duplex voice frontend that delegates to Hermes. + +``voice.voice_chat_mode: gpt-live`` replaces the chained STT → turn → TTS loop with ONE +full-duplex voice model (OpenAI ``gpt-live-1``) that owns the microphone and the speaker and +delegates every real request to Hermes as its *client-delegation* backend. Hermes stays the +agent: whatever model/provider the session has selected answers, with the full toolset. + +Division of labour (the Live API has no tools of its own in client mode): + +* the desktop renderer holds the WebRTC media session (mic in, speech out) and the data channel; +* this module resolves WHICH credentials/voice/persona to use and performs the one server-side + step the API requires — exchanging the browser's SDP offer for an answer with the project key + (``POST /v1/live/sessions``), so the key never reaches the client; +* the renderer turns each ``session.delegation.created`` into a normal ``prompt.submit`` on the + active session (surface ``voice-live``) and streams the reply back as + ``session.commentary.append`` — Hermes' answer is what the voice speaks. + +Vendor contract: https://developers.openai.com/api/docs/guides/live (+ live-delegation, +voice-webrtc). Billing is $0.05/min of session time on the OpenAI key, separate from the +Hermes turn. +""" + +from __future__ import annotations + +import json +import logging +import urllib.error +import urllib.request +from typing import Any, Dict, Optional + +logger = logging.getLogger(__name__) + +GPT_LIVE_MODE = "gpt-live" +CHAINED_MODE = "chained" +DEFAULT_LIVE_MODEL = "gpt-live-1" +DEFAULT_LIVE_VOICE = "marin" +DEFAULT_LIVE_BASE_URL = "https://api.openai.com/v1" +# Voices the vendor lists for gpt-live-1 (live-conversations guide) plus the realtime defaults it +# accepts; free text stays allowed for custom voices. +GPT_LIVE_VOICES = ( + "marin", "cedar", "quartz", "ripple", "vesper", "willow", "stone", "gleam", "meridian", + "bossa", "tempo", "beacon", "delta", "cinder", +) + +# Persona for the voice layer. Short on purpose: the live model has a small context window and +# the vendor guide asks for role + style + a labelled delegation policy, nothing more. The +# backend (Hermes) carries the real instructions, tools and memory. +LIVE_PERSONA = ( + "You are Hermes, a calm and friendly voice assistant. Speak naturally at an unhurried pace. " + "Be clear and direct, not overly cheerful. If the user is frustrated, acknowledge it briefly " + "and focus on the next helpful step.\n\n" + "Backchannel policy: Use moderate backchannels. Acknowledge naturally without competing with " + "the main response.\n\n" + "Interruption policy: Stop speaking when the user interrupts. Listen to what they say.\n\n" + "Delegation policy:\n" + "Backend tools:\n" + "- Hermes agent: a full AI agent with tools — it can run commands, read and edit files, " + "browse the web, search, remember things across sessions, schedule tasks, and reason " + "carefully about anything. It is the one who actually does work and knows facts.\n\n" + "Delegate to the backend when:\n" + "- The user asks a question that needs facts, current information, or careful reasoning.\n" + "- The user asks you to do, check, find, make, fix, run or remember anything.\n" + "- A correction changes work already requested.\n\n" + "Do not delegate to the backend when:\n" + "- The user greets you, makes small talk, or asks you to repeat a result already provided.\n" + "- You need a brief clarification to understand the request.\n\n" + "Delegate before giving an answer that depends on backend work. Do not guess the result " + "while waiting; say briefly that you are checking, then wait for the result." +) + +# Per-turn note prepended to the MODEL INPUT (never the byte-stable system prompt) when a turn +# arrives from the live voice layer. Same seam as the HUD note. +VOICE_LIVE_TURN_NOTE = ( + "[Note: this message is a delegation from a live spoken conversation. The text is a voice " + "transcript (it may contain mis-hearings, hesitations and later corrections; use the latest " + "intent). Your reply will be spoken aloud by a voice model that paraphrases it: answer in plain " + "conversational sentences, keep it short (a few sentences unless the user asked for detail), no " + "markdown, no lists, no code blocks, no URLs read out character by character. Do the work with " + "your tools as usual; only the final facts need to be spoken. Do not claim an action succeeded " + "before it actually did.]" +) + + +def voice_live_turn_note(context: str = "") -> str: + """The per-turn note plus, when the client sent one, the recent spoken exchange the delegation + refers to (the user's last words alone are often "yes" or "Thursday, not Friday").""" + context = context.strip() + if not context: + return VOICE_LIVE_TURN_NOTE + return f"{VOICE_LIVE_TURN_NOTE}\n[Recent spoken conversation, newest last:\n{context}]" + + +def _voice_section() -> Dict[str, Any]: + try: + from hermes_cli.config import load_config + voice = load_config().get("voice") + except Exception: + return {} + return voice if isinstance(voice, dict) else {} + + +def _live_section(voice: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: + section = (voice if voice is not None else _voice_section()).get("gpt_live") + return section if isinstance(section, dict) else {} + + +def voice_chat_mode(voice: Optional[Dict[str, Any]] = None) -> str: + """``chained`` (default) or ``gpt-live``. Accepts the underscore spelling too.""" + raw = (voice if voice is not None else _voice_section()).get("voice_chat_mode") + mode = str(raw or CHAINED_MODE).strip().lower().replace("_", "-") + return GPT_LIVE_MODE if mode in {GPT_LIVE_MODE, "gptlive", "live"} else CHAINED_MODE + + +def _resolve_credentials(live: Dict[str, Any]) -> tuple[str, str]: + """``(api_key, base_url)`` — ``voice.gpt_live.api_key`` first, else the same OpenAI audio + chain the STT/TTS providers use (``VOICE_TOOLS_OPENAI_KEY`` → ``OPENAI_API_KEY`` → pool). + + The Nous-managed audio proxy does not carry ``/live/sessions``; this mode is direct-key only. + """ + from tools.tool_backend_helpers import resolve_openai_audio_api_key + api_key = str(live.get("api_key") or "").strip() or resolve_openai_audio_api_key() + base_url = str(live.get("base_url") or DEFAULT_LIVE_BASE_URL).strip().rstrip("/") + return api_key, base_url + + +def live_instructions(live: Optional[Dict[str, Any]] = None) -> str: + extra = str((live if live is not None else _live_section()).get("instructions") or "").strip() + return f"{LIVE_PERSONA}\n\n{extra}" if extra else LIVE_PERSONA + + +def resolve_gpt_live_status() -> Dict[str, Any]: + """Non-secret readiness verdict for the client: which mode is selected and whether GPT-Live + can start (a key resolves). Never returns the key.""" + voice = _voice_section() + mode = voice_chat_mode(voice) + live = _live_section(voice) + api_key, _base = _resolve_credentials(live) + return { + "mode": mode, + "available": bool(api_key), + "reason": None if api_key else "no OpenAI API key (set OPENAI_API_KEY or voice.gpt_live.api_key)", + "model": str(live.get("model") or DEFAULT_LIVE_MODEL), + "voice": str(live.get("voice") or DEFAULT_LIVE_VOICE), + } + + +def build_session_config(history: Optional[list] = None) -> Dict[str, Any]: + """The ``session`` object for ``POST /v1/live/sessions`` (client delegation, WebRTC — the + transport negotiates the audio format, so none is set).""" + live = _live_section() + config: Dict[str, Any] = { + "model": str(live.get("model") or DEFAULT_LIVE_MODEL), + "instructions": live_instructions(live), + "audio": {"output": {"voice": str(live.get("voice") or DEFAULT_LIVE_VOICE)}}, + "delegation": {"type": "client"}, + } + if history: + config["input"] = history + return config + + +def create_webrtc_session(sdp_offer: str, history: Optional[list] = None) -> Dict[str, Any]: + """Exchange the renderer's SDP offer for the Live session answer. + + Returns the vendor response ``{"session": {"id": ...}, "transport": {"type": "webrtc", + "sdp": ...}}``. Raises ``ValueError`` for a missing key and ``RuntimeError`` (with the vendor + status/detail) for a rejected request. + """ + live = _live_section() + api_key, base_url = _resolve_credentials(live) + if not api_key: + raise ValueError("GPT-Live needs an OpenAI API key (OPENAI_API_KEY or voice.gpt_live.api_key)") + body = json.dumps({ + "session": build_session_config(history), + "transport": {"type": "webrtc", "sdp": sdp_offer}, + }).encode("utf-8") + req = urllib.request.Request( + f"{base_url}/live/sessions", data=body, method="POST", + headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"}) + try: + with urllib.request.urlopen(req, timeout=30) as resp: + return json.loads(resp.read().decode("utf-8")) + except urllib.error.HTTPError as exc: + detail = exc.read().decode("utf-8", "replace")[:600] + logger.warning("GPT-Live session creation failed: %s %s", exc.code, detail) + raise RuntimeError(f"GPT-Live session creation failed ({exc.code}): {detail}") from exc diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index 0ff67007c2..bdb0a40b92 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -537,6 +537,10 @@ def _lock_in_submit_turn( return None, fields +# Per-turn client surfaces that carry a model-bound note (session_notifications._surface_note). +_CLIENT_SURFACES = frozenset({"hud", "voice-live"}) + + @method("prompt.submit") def _(rid, params: dict) -> dict: from hermes_cli.input_sanitize import sanitize_user_prompt_text @@ -575,8 +579,13 @@ def _(rid, params: dict) -> dict: # leaves the session untouched. The reason travels as machine-readable data. reason = getattr(limit_message, "reason", None) return _err(rid, 4090, str(limit_message), {"reason": reason} if reason else None) - # Rewritten every submit: a session alternates app window / HUD; stale "hud" misinforms. - session["client_surface"] = "hud" if params.get("surface") == "hud" else "" + # Rewritten every submit: a session alternates app window / HUD / live voice; a stale value misinforms. + session["client_surface"] = params.get("surface") if params.get("surface") in _CLIENT_SURFACES else "" + # Live-voice delegations carry the recent spoken transcript for the MODEL INPUT only (the persisted + # user row stays the words the user said); anything else clears it. + voice_context = params.get("voice_context") + session["voice_live_context"] = ( + voice_context[:6000] if session["client_surface"] == "voice-live" and isinstance(voice_context, str) else "") has_truncation = any(params.get(k) is not None for k in _TRUNCATION_PARAMS) if has_truncation and isinstance(text, str): # A rewind replays what the transcript shows: re-expand a skill invocation or diff --git a/tui_gateway/session_notifications.py b/tui_gateway/session_notifications.py index c6d6368f53..5d16abf471 100644 --- a/tui_gateway/session_notifications.py +++ b/tui_gateway/session_notifications.py @@ -662,11 +662,16 @@ def _start_notification_poller(sid: str, session: dict) -> threading.Event: def _hud_surface_note(session: dict) -> str: - """The HUD-mode note for this turn, or "" when it was not typed there.""" - if session.get("client_surface") != "hud": - return "" - from agent.prompt_builder import hud_surface_note - return hud_surface_note(getattr(session.get("agent"), "valid_tool_names", None)) + """The per-surface note for this turn ("" for the plain app window): HUD → the read-the-window-below + prior; voice-live → the spoken-delegation contract (transcript in, speakable prose out).""" + surface = session.get("client_surface") + if surface == "hud": + from agent.prompt_builder import hud_surface_note + return hud_surface_note(getattr(session.get("agent"), "valid_tool_names", None)) + if surface == "voice-live": + from tools.voice_live import voice_live_turn_note + return voice_live_turn_note(session.get("voice_live_context") or "") + return "" def _prepend_note(run_message: Any, note: str) -> Any: diff --git a/website/docs/user-guide/features/voice-mode.md b/website/docs/user-guide/features/voice-mode.md index 349b8936d2..8750029e2b 100644 --- a/website/docs/user-guide/features/voice-mode.md +++ b/website/docs/user-guide/features/voice-mode.md @@ -193,6 +193,24 @@ voice: Client-direct wire support: OpenAI (incl. Nous-managed audio), Groq, Mistral, and DeepInfra via the OpenAI-compatible shapes, xAI Grok STT, and ElevenLabs STT + TTS. xAI configured through OAuth stays on the relay (the OAuth bearer refreshes server-side). +### Desktop: GPT-Live voice chat mode (full duplex, delegates to Hermes) + +The chained loop above is one of two voice chat modes in the desktop app. The other replaces the whole STT → turn → TTS chain with **one full-duplex voice model**, OpenAI's `gpt-live-1`: it listens while it speaks, handles interruptions, backchannels and background noise itself, and has **no tools of its own**. Whenever you ask for real work it *delegates* to Hermes, which answers as usual — with whatever model and provider the session has selected, the full toolset, memory and approvals — and the voice paraphrases the answer aloud. + +```yaml +voice: + voice_chat_mode: gpt-live # chained (default) | gpt-live + gpt_live: + voice: marin # marin, cedar, quartz, ripple, vesper, willow, stone, gleam, meridian, … + instructions: "" # optional extra persona sentences (tone, pace, language) +``` + +Requirements: an OpenAI API key (`OPENAI_API_KEY`, `VOICE_TOOLS_OPENAI_KEY`, or `voice.gpt_live.api_key`). The voice layer is billed by OpenAI at **$0.05 per minute of session time** (idle time counts); the Hermes turn is billed on its own provider as always. The mode is also in Settings → Voice → *Voice Chat Mode*. + +How it works: pressing the voice button opens a WebRTC session from the desktop to GPT-Live; the desktop only ever receives a session id and an SDP answer — the key stays on the gateway host, which performs the session creation (`POST /api/audio/voice-live/session`). Each `session.delegation.created` becomes a normal turn on the open chat (the bubble shows what you said; the recent spoken exchange rides the model input as a per-turn note, never the system prompt, so the reply is speakable prose). Tool activity is fed to the voice as quiet context ("Hermes is working: terminal") so it can tell you what is happening if you ask; the final answer is streamed back sentence by sentence. Saying the stop phrase ends the conversation. If `gpt-live` is selected but no key resolves, the button falls back to the chained mode with a notice. + +Not supported in this mode: the Nous-managed audio proxy (direct key only), the CLI/TUI (`/voice` keeps the chained loop), and the `tts` tool (it keeps using `tts.provider`). + ### Barge-in You can interrupt the agent at ANY point in its turn — the microphone stays live from the moment you finish speaking until the reply has fully played (full duplex):