fix(desktop): keep read-aloud independent of gateway auto-TTS
Salvage #99095 (e7ea53074ab2b64a1530641659399d9f1bb4b435), completing one-time migration for both boolean values and using the existing storage helpers. Hydration and local toggles never edit backend configuration. Fixes #99076.
This commit is contained in:
@@ -323,7 +323,7 @@ function ConversationIndicator({
|
||||
|
||||
// Pure-TTS toggle: type normally, but have every assistant reply read aloud —
|
||||
// no dictation, no full conversation loop. Filled/accent when on, mirroring the
|
||||
// muted-mic pressed state above. Driven by (and persisted to) `voice.auto_tts`.
|
||||
// muted-mic pressed state above. Persisted locally, independently of gateway TTS.
|
||||
function AutoSpeakButton({ active, disabled, onToggle }: { active: boolean; disabled: boolean; onToggle: () => void }) {
|
||||
const { t } = useI18n()
|
||||
const c = t.composer
|
||||
|
||||
@@ -5,7 +5,32 @@ vi.mock('@/hermes', () => ({
|
||||
saveHermesConfig: vi.fn(async () => undefined)
|
||||
}))
|
||||
|
||||
import { $voiceStopPhrase, applyVoiceStopPhraseFromConfig } from './voice-prefs'
|
||||
import { saveHermesConfig } from '@/hermes'
|
||||
|
||||
import { $autoSpeakReplies, $voiceStopPhrase, applyAutoSpeakFromConfig, applyVoiceStopPhraseFromConfig, setAutoSpeakReplies } from './voice-prefs'
|
||||
|
||||
it('keeps the desktop toggle local across config refreshes', async () => {
|
||||
localStorage.clear()
|
||||
$autoSpeakReplies.set(false)
|
||||
vi.mocked(saveHermesConfig).mockClear()
|
||||
await setAutoSpeakReplies(true)
|
||||
applyAutoSpeakFromConfig({ voice: { auto_tts: false } })
|
||||
expect($autoSpeakReplies.get()).toBe(true)
|
||||
expect(saveHermesConfig).not.toHaveBeenCalled()
|
||||
expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBe('true')
|
||||
})
|
||||
|
||||
it('migrates the legacy preference once, not on every refresh', () => {
|
||||
for (const enabled of [false, true]) {
|
||||
localStorage.clear()
|
||||
applyAutoSpeakFromConfig(null)
|
||||
expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBeNull()
|
||||
applyAutoSpeakFromConfig({ voice: { auto_tts: enabled } })
|
||||
applyAutoSpeakFromConfig({ voice: { auto_tts: !enabled } })
|
||||
expect($autoSpeakReplies.get()).toBe(enabled)
|
||||
expect(localStorage.getItem('hermes.desktop.autoSpeakReplies')).toBe(String(enabled))
|
||||
}
|
||||
})
|
||||
|
||||
describe('applyVoiceStopPhraseFromConfig', () => {
|
||||
it('defaults to "stop" when the key is absent (backend default applies)', () => {
|
||||
|
||||
@@ -1,15 +1,16 @@
|
||||
import { atom } from 'nanostores'
|
||||
|
||||
import { getHermesConfigRecord, saveHermesConfig } from '@/hermes'
|
||||
import { persistBoolean, readKey, storedBoolean } from '@/lib/storage'
|
||||
|
||||
// "Read replies aloud" — mirrors the canonical `voice.auto_tts` config key (also
|
||||
// in Settings → Voice, honored by the messaging gateway) so the composer toggle
|
||||
// and the Settings switch are one source of truth, not two that can disagree.
|
||||
export const $autoSpeakReplies = atom<boolean>(false)
|
||||
// Desktop read-aloud is local; voice.auto_tts belongs to the messaging gateway.
|
||||
const AUTO_SPEAK_KEY = 'hermes.desktop.autoSpeakReplies'
|
||||
export const $autoSpeakReplies = atom<boolean>(storedBoolean(AUTO_SPEAK_KEY, false))
|
||||
|
||||
/** Seed the atom from a loaded config payload (mount / refresh). */
|
||||
/** Migrate the legacy value once without editing the backend configuration. */
|
||||
export function applyAutoSpeakFromConfig(config: { voice?: { auto_tts?: unknown } | null } | null | undefined) {
|
||||
$autoSpeakReplies.set(Boolean(config?.voice?.auto_tts))
|
||||
if (config != null && readKey(AUTO_SPEAK_KEY) === null) {
|
||||
void setAutoSpeakReplies(Boolean(config.voice?.auto_tts))
|
||||
}
|
||||
}
|
||||
|
||||
// First configured `voice.stop_phrases` entry — drives the "Say "stop" to end
|
||||
@@ -48,27 +49,8 @@ export function applyThinkingSoundFromConfig(
|
||||
$thinkingSoundEnabled.set(config?.voice?.thinking_sound !== false)
|
||||
}
|
||||
|
||||
/**
|
||||
* Flip the preference and persist it. Optimistic — the atom updates instantly and
|
||||
* reverts if the config write fails. Read-modify-writes the whole record (the
|
||||
* same path the Settings page uses; there's no partial-update endpoint).
|
||||
*/
|
||||
/** Persist even an unchanged value, so migrating false is also one-time. */
|
||||
export async function setAutoSpeakReplies(enabled: boolean): Promise<void> {
|
||||
const previous = $autoSpeakReplies.get()
|
||||
|
||||
if (previous === enabled) {
|
||||
return
|
||||
}
|
||||
|
||||
persistBoolean(AUTO_SPEAK_KEY, enabled)
|
||||
$autoSpeakReplies.set(enabled)
|
||||
|
||||
try {
|
||||
const record = await getHermesConfigRecord()
|
||||
const voice = record.voice && typeof record.voice === 'object' ? (record.voice as Record<string, unknown>) : {}
|
||||
|
||||
await saveHermesConfig({ ...record, voice: { ...voice, auto_tts: enabled } })
|
||||
} catch (error) {
|
||||
$autoSpeakReplies.set(previous)
|
||||
throw error
|
||||
}
|
||||
}
|
||||
|
||||
@@ -260,7 +260,7 @@ tts:
|
||||
|
||||
Local engines (Piper, KittenTTS) load their model lazily, so without help the *first* spoken reply after you turn speech on pays the whole model load — and on a fresh install the voice download — as silence before the first word. Hermes treats the speech-output toggles as the signal that TTS is about to be needed:
|
||||
|
||||
- **Desktop** — turning on **Read replies aloud**, or starting a **voice conversation**, pre-loads the configured engine in the background right away. Turning both off again unloads the resident model (a Piper voice is tens of MB; KittenTTS up to ~80MB) so it isn't parked in RAM for nothing.
|
||||
- **Desktop** — **Read replies aloud** is a desktop-local preference, independent of the gateway's `voice.auto_tts` setting in Settings → Voice. It migrates the shared value once, then later gateway configuration changes do not override the desktop toggle. Turning on **Read replies aloud**, or starting a **voice conversation**, pre-loads the configured engine in the background right away. Turning both off again unloads the resident model (a Piper voice is tens of MB; KittenTTS up to ~80MB) so it isn't parked in RAM for nothing.
|
||||
- **CLI / TUI** — `/voice tts` (and `/voice on` when `voice.auto_tts` is set) do the same; `/voice off` releases.
|
||||
|
||||
Each toggle holds a *lease* on the engine; the model is only unloaded when the last lease across surfaces is released, so switching off read-aloud in one Desktop window never pulls the voice out from under a conversation running in another. For cloud providers there is no model to hold — the toggle only makes sure a lazily-installed SDK (edge-tts, ElevenLabs, Mistral) is present. Warm-up is best-effort: if the engine can't load, the toggle still succeeds and the first reply falls back to loading on demand as before.
|
||||
|
||||
Reference in New Issue
Block a user