diff --git a/apps/desktop/src/app/chat/composer/controls.tsx b/apps/desktop/src/app/chat/composer/controls.tsx index 388df3cdc0..51ea823427 100644 --- a/apps/desktop/src/app/chat/composer/controls.tsx +++ b/apps/desktop/src/app/chat/composer/controls.tsx @@ -323,7 +323,7 @@ function ConversationIndicator({ // Pure-TTS toggle: type normally, but have every assistant reply read aloud — // no dictation, no full conversation loop. Filled/accent when on, mirroring the -// muted-mic pressed state above. Driven by (and persisted to) `voice.auto_tts`. +// muted-mic pressed state above. Persisted locally, independently of gateway TTS. function AutoSpeakButton({ active, disabled, onToggle }: { active: boolean; disabled: boolean; onToggle: () => void }) { const { t } = useI18n() const c = t.composer diff --git a/apps/desktop/src/store/voice-prefs.test.ts b/apps/desktop/src/store/voice-prefs.test.ts index e648aa744b..76227f2dc5 100644 --- a/apps/desktop/src/store/voice-prefs.test.ts +++ b/apps/desktop/src/store/voice-prefs.test.ts @@ -5,7 +5,32 @@ vi.mock('@/hermes', () => ({ saveHermesConfig: vi.fn(async () => undefined) })) -import { $voiceStopPhrase, applyVoiceStopPhraseFromConfig } from './voice-prefs' +import { saveHermesConfig } from '@/hermes' + +import { $autoSpeakReplies, $voiceStopPhrase, applyAutoSpeakFromConfig, applyVoiceStopPhraseFromConfig, setAutoSpeakReplies } from './voice-prefs' + +it('keeps the desktop toggle local across config refreshes', async () => { + localStorage.clear() + $autoSpeakReplies.set(false) + vi.mocked(saveHermesConfig).mockClear() + await setAutoSpeakReplies(true) + applyAutoSpeakFromConfig({ voice: { auto_tts: false } }) + expect($autoSpeakReplies.get()).toBe(true) + expect(saveHermesConfig).not.toHaveBeenCalled() + expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBe('true') +}) + +it('migrates the legacy preference once, not on every refresh', () => { + for (const enabled of [false, true]) { + localStorage.clear() + applyAutoSpeakFromConfig(null) + expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBeNull() + applyAutoSpeakFromConfig({ voice: { auto_tts: enabled } }) + applyAutoSpeakFromConfig({ voice: { auto_tts: !enabled } }) + expect($autoSpeakReplies.get()).toBe(enabled) + expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBe(String(enabled)) + } +}) describe('applyVoiceStopPhraseFromConfig', () => { it('defaults to "stop" when the key is absent (backend default applies)', () => { diff --git a/apps/desktop/src/store/voice-prefs.ts b/apps/desktop/src/store/voice-prefs.ts index 51a073c23b..fdb13873ec 100644 --- a/apps/desktop/src/store/voice-prefs.ts +++ b/apps/desktop/src/store/voice-prefs.ts @@ -1,15 +1,16 @@ import { atom } from 'nanostores' -import { getHermesConfigRecord, saveHermesConfig } from '@/hermes' +import { persistBoolean, readKey, storedBoolean } from '@/lib/storage' -// "Read replies aloud" — mirrors the canonical `voice.auto_tts` config key (also -// in Settings → Voice, honored by the messaging gateway) so the composer toggle -// and the Settings switch are one source of truth, not two that can disagree. -export const $autoSpeakReplies = atom(false) +// Desktop read-aloud is local; voice.auto_tts belongs to the messaging gateway. +const AUTO_SPEAK_KEY = 'hermes.desktop.autoSpeakReplies' +export const $autoSpeakReplies = atom(storedBoolean(AUTO_SPEAK_KEY, false)) -/** Seed the atom from a loaded config payload (mount / refresh). */ +/** Migrate the legacy value once without editing the backend configuration. */ export function applyAutoSpeakFromConfig(config: { voice?: { auto_tts?: unknown } | null } | null | undefined) { - $autoSpeakReplies.set(Boolean(config?.voice?.auto_tts)) + if (config != null && readKey(AUTO_SPEAK_KEY) === null) { + void setAutoSpeakReplies(Boolean(config.voice?.auto_tts)) + } } // First configured `voice.stop_phrases` entry — drives the "Say "stop" to end @@ -48,27 +49,8 @@ export function applyThinkingSoundFromConfig( $thinkingSoundEnabled.set(config?.voice?.thinking_sound !== false) } -/** - * Flip the preference and persist it. Optimistic — the atom updates instantly and - * reverts if the config write fails. Read-modify-writes the whole record (the - * same path the Settings page uses; there's no partial-update endpoint). - */ +/** Persist even an unchanged value, so migrating false is also one-time. */ export async function setAutoSpeakReplies(enabled: boolean): Promise { - const previous = $autoSpeakReplies.get() - - if (previous === enabled) { - return - } - + persistBoolean(AUTO_SPEAK_KEY, enabled) $autoSpeakReplies.set(enabled) - - try { - const record = await getHermesConfigRecord() - const voice = record.voice && typeof record.voice === 'object' ? (record.voice as Record) : {} - - await saveHermesConfig({ ...record, voice: { ...voice, auto_tts: enabled } }) - } catch (error) { - $autoSpeakReplies.set(previous) - throw error - } } diff --git a/website/docs/user-guide/features/tts.md b/website/docs/user-guide/features/tts.md index e2ae021a81..7261b27e50 100644 --- a/website/docs/user-guide/features/tts.md +++ b/website/docs/user-guide/features/tts.md @@ -260,7 +260,7 @@ tts: Local engines (Piper, KittenTTS) load their model lazily, so without help the *first* spoken reply after you turn speech on pays the whole model load — and on a fresh install the voice download — as silence before the first word. Hermes treats the speech-output toggles as the signal that TTS is about to be needed: -- **Desktop** — turning on **Read replies aloud**, or starting a **voice conversation**, pre-loads the configured engine in the background right away. Turning both off again unloads the resident model (a Piper voice is tens of MB; KittenTTS up to ~80MB) so it isn't parked in RAM for nothing. +- **Desktop** — **Read replies aloud** is a desktop-local preference, independent of the gateway's `voice.auto_tts` setting in Settings → Voice. It migrates the shared value once, then later gateway configuration changes do not override the desktop toggle. Turning on **Read replies aloud**, or starting a **voice conversation**, pre-loads the configured engine in the background right away. Turning both off again unloads the resident model (a Piper voice is tens of MB; KittenTTS up to ~80MB) so it isn't parked in RAM for nothing. - **CLI / TUI** — `/voice tts` (and `/voice on` when `voice.auto_tts` is set) do the same; `/voice off` releases. Each toggle holds a *lease* on the engine; the model is only unloaded when the last lease across surfaces is released, so switching off read-aloud in one Desktop window never pulls the voice out from under a conversation running in another. For cloud providers there is no model to hold — the toggle only makes sure a lazily-installed SDK (edge-tts, ElevenLabs, Mistral) is present. Warm-up is best-effort: if the engine can't load, the toggle still succeeds and the first reply falls back to loading on demand as before.