fix(desktop): keep read-aloud independent of gateway auto-TTS

Salvage #99095 (e7ea53074ab2b64a1530641659399d9f1bb4b435), completing one-time migration for both boolean values and using the existing storage helpers. Hydration and local toggles never edit backend configuration. Fixes #99076.
This commit is contained in:
liuhao1024
2026-09-07 01:01:22 -07:00
committed by Teknium
parent 44a583fcc8
commit 938a3dc8f9
4 changed files with 38 additions and 31 deletions
@@ -323,7 +323,7 @@ function ConversationIndicator({
// Pure-TTS toggle: type normally, but have every assistant reply read aloud —
// no dictation, no full conversation loop. Filled/accent when on, mirroring the
// muted-mic pressed state above. Driven by (and persisted to) `voice.auto_tts`.
// muted-mic pressed state above. Persisted locally, independently of gateway TTS.
function AutoSpeakButton({ active, disabled, onToggle }: { active: boolean; disabled: boolean; onToggle: () => void }) {
const { t } = useI18n()
const c = t.composer
+26 -1
View File
@@ -5,7 +5,32 @@ vi.mock('@/hermes', () => ({
saveHermesConfig: vi.fn(async () => undefined)
}))
import { $voiceStopPhrase, applyVoiceStopPhraseFromConfig } from './voice-prefs'
import { saveHermesConfig } from '@/hermes'
import { $autoSpeakReplies, $voiceStopPhrase, applyAutoSpeakFromConfig, applyVoiceStopPhraseFromConfig, setAutoSpeakReplies } from './voice-prefs'
it('keeps the desktop toggle local across config refreshes', async () => {
localStorage.clear()
$autoSpeakReplies.set(false)
vi.mocked(saveHermesConfig).mockClear()
await setAutoSpeakReplies(true)
applyAutoSpeakFromConfig({ voice: { auto_tts: false } })
expect($autoSpeakReplies.get()).toBe(true)
expect(saveHermesConfig).not.toHaveBeenCalled()
expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBe('true')
})
it('migrates the legacy preference once, not on every refresh', () => {
for (const enabled of [false, true]) {
localStorage.clear()
applyAutoSpeakFromConfig(null)
expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBeNull()
applyAutoSpeakFromConfig({ voice: { auto_tts: enabled } })
applyAutoSpeakFromConfig({ voice: { auto_tts: !enabled } })
expect($autoSpeakReplies.get()).toBe(enabled)
expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBe(String(enabled))
}
})
describe('applyVoiceStopPhraseFromConfig', () => {
it('defaults to "stop" when the key is absent (backend default applies)', () => {
+10 -28
View File
@@ -1,15 +1,16 @@
import { atom } from 'nanostores'
import { getHermesConfigRecord, saveHermesConfig } from '@/hermes'
import { persistBoolean, readKey, storedBoolean } from '@/lib/storage'
// "Read replies aloud" — mirrors the canonical `voice.auto_tts` config key (also
// in Settings → Voice, honored by the messaging gateway) so the composer toggle
// and the Settings switch are one source of truth, not two that can disagree.
export const $autoSpeakReplies = atom<boolean>(false)
// Desktop read-aloud is local; voice.auto_tts belongs to the messaging gateway.
const AUTO_SPEAK_KEY = 'hermes.desktop.autoSpeakReplies'
export const $autoSpeakReplies = atom<boolean>(storedBoolean(AUTO_SPEAK_KEY, false))
/** Seed the atom from a loaded config payload (mount / refresh). */
/** Migrate the legacy value once without editing the backend configuration. */
export function applyAutoSpeakFromConfig(config: { voice?: { auto_tts?: unknown } | null } | null | undefined) {
$autoSpeakReplies.set(Boolean(config?.voice?.auto_tts))
if (config != null && readKey(AUTO_SPEAK_KEY) === null) {
void setAutoSpeakReplies(Boolean(config.voice?.auto_tts))
}
}
// First configured `voice.stop_phrases` entry — drives the "Say "stop" to end
@@ -48,27 +49,8 @@ export function applyThinkingSoundFromConfig(
$thinkingSoundEnabled.set(config?.voice?.thinking_sound !== false)
}
/**
* Flip the preference and persist it. Optimistic — the atom updates instantly and
* reverts if the config write fails. Read-modify-writes the whole record (the
* same path the Settings page uses; there's no partial-update endpoint).
*/
/** Persist even an unchanged value, so migrating false is also one-time. */
export async function setAutoSpeakReplies(enabled: boolean): Promise<void> {
const previous = $autoSpeakReplies.get()
if (previous === enabled) {
return
}
persistBoolean(AUTO_SPEAK_KEY, enabled)
$autoSpeakReplies.set(enabled)
try {
const record = await getHermesConfigRecord()
const voice = record.voice && typeof record.voice === 'object' ? (record.voice as Record<string, unknown>) : {}
await saveHermesConfig({ ...record, voice: { ...voice, auto_tts: enabled } })
} catch (error) {
$autoSpeakReplies.set(previous)
throw error
}
}
+1 -1
View File
@@ -260,7 +260,7 @@ tts:
Local engines (Piper, KittenTTS) load their model lazily, so without help the *first* spoken reply after you turn speech on pays the whole model load — and on a fresh install the voice download — as silence before the first word. Hermes treats the speech-output toggles as the signal that TTS is about to be needed:
- **Desktop** — turning on **Read replies aloud**, or starting a **voice conversation**, pre-loads the configured engine in the background right away. Turning both off again unloads the resident model (a Piper voice is tens of MB; KittenTTS up to ~80MB) so it isn't parked in RAM for nothing.
- **Desktop** — **Read replies aloud** is a desktop-local preference, independent of the gateway's `voice.auto_tts` setting in Settings → Voice. It migrates the shared value once, then later gateway configuration changes do not override the desktop toggle. Turning on **Read replies aloud**, or starting a **voice conversation**, pre-loads the configured engine in the background right away. Turning both off again unloads the resident model (a Piper voice is tens of MB; KittenTTS up to ~80MB) so it isn't parked in RAM for nothing.
- **CLI / TUI** — `/voice tts` (and `/voice on` when `voice.auto_tts` is set) do the same; `/voice off` releases.
Each toggle holds a *lease* on the engine; the model is only unloaded when the last lease across surfaces is released, so switching off read-aloud in one Desktop window never pulls the voice out from under a conversation running in another. For cloud providers there is no model to hold — the toggle only makes sure a lazily-installed SDK (edge-tts, ElevenLabs, Mistral) is present. Warm-up is best-effort: if the engine can't load, the toggle still succeeds and the first reply falls back to loading on demand as before.