From 782dd635fe0cac38e66656dbee0ffa554403da55 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:37:13 -0700 Subject: [PATCH] feat(tts): speech toggles warm/release plugin and command TTS providers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extends the TTS lease from #100912 (4e3feb8bbb0) beyond built-in local engines: when the configured tts.provider is user-declared, acquiring the first lease and releasing the last one now reach it, so a self-hosted TTS server can preload its model when read-aloud / voice conversation turns on and unload when it turns off (Discord request). - agent/tts_provider.py: TTSProvider gains concrete no-op warm() / release() (not abstract — existing plugins are unaffected). - tools/tts_tool.py: _signal_user_tts_provider() forwards the lease hook; plugin providers get warm()/release(), command providers run optional `warm_command` / `release_command` (config.yaml, under tts.providers.) through the existing _run_command_tts helper on a daemon thread — best-effort, output discarded, failures at debug. warm_tts_provider() and release_tts_provider() call it. - tests/tools/test_tts_lifecycle_leases.py: fake plugin provider and fake command provider observe warm/release through acquire/release lease (both fail on main with action == "noop"). - docs: features/tts.md — lease section, command-provider optional keys table, plugin optional hooks. --- agent/tts_provider.py | 16 ++++++ tests/tools/test_tts_lifecycle_leases.py | 72 ++++++++++++++++++++++++ tools/tts_tool.py | 60 +++++++++++++++++++- website/docs/user-guide/features/tts.md | 4 ++ 4 files changed, 151 insertions(+), 1 deletion(-) diff --git a/agent/tts_provider.py b/agent/tts_provider.py index c19166a702..075cab3f40 100644 --- a/agent/tts_provider.py +++ b/agent/tts_provider.py @@ -241,6 +241,22 @@ class TTSProvider(abc.ABC): "if your backend supports it." ) + def warm(self) -> None: + """Speech output was just turned on; pre-load so the first reply is hot. + + Optional. Called from the TTS lease path (Desktop read-aloud / voice + conversation, ``/voice tts``) when this provider is the configured + ``tts.provider`` — e.g. ask a local model server to load its model. + Best-effort: exceptions are logged at debug and ignored. Default: no-op. + """ + + def release(self) -> None: + """The last speech-output lease was released; free resident resources. + + Optional counterpart of :meth:`warm` — e.g. tell a local model server + to unload. Best-effort; default: no-op. + """ + @property def voice_compatible(self) -> bool: """Whether output is suitable for voice-bubble delivery. diff --git a/tests/tools/test_tts_lifecycle_leases.py b/tests/tools/test_tts_lifecycle_leases.py index e84478dbff..7558848f8e 100644 --- a/tests/tools/test_tts_lifecycle_leases.py +++ b/tests/tools/test_tts_lifecycle_leases.py @@ -9,6 +9,8 @@ resident local models. from __future__ import annotations +import threading + import pytest from tools import tts_tool @@ -231,3 +233,73 @@ def test_every_local_warmer_has_a_registered_cache(): assert set(warmers) == set(tts_tool._LOCAL_TTS_MODEL_CACHES) assert tts_tool._LOCAL_TTS_MODEL_CACHES["piper"] is tts_tool._piper_voice_cache assert tts_tool._LOCAL_TTS_MODEL_CACHES["kittentts"] is tts_tool._kittentts_model_cache + + +# -------------------------------------------------------------------------- +# User-declared providers get the same signal (plugin warm()/release(), +# command warm_command/release_command) so a local TTS server can preload +# and unload on the speech toggles. +# -------------------------------------------------------------------------- + + +def test_plugin_provider_warm_and_release_follow_the_lease(monkeypatch): + from agent import tts_provider, tts_registry + + calls: list = [] + + class _ServerBacked(tts_provider.TTSProvider): + @property + def name(self): + return "my-server" + + def synthesize(self, text, output_path, **kw): + return output_path + + def warm(self): + calls.append("warm") + + def release(self): + calls.append("release") + + tts_registry._reset_for_tests() + tts_registry.register_provider(_ServerBacked()) + cfg = {"provider": "my-server"} + monkeypatch.setattr(tts_tool, "_load_tts_config", lambda: cfg) + monkeypatch.setattr("hermes_cli.plugins._ensure_plugins_discovered", lambda force=False: None) + try: + assert tts_tool.acquire_tts_lease("desktop:read-aloud", cfg)["action"] == "warmed" + tts_tool.acquire_tts_lease("tui:voice-tts", cfg) + tts_tool.release_tts_lease("desktop:read-aloud") + assert calls == ["warm", "warm"] # still one holder — no release yet + tts_tool.release_tts_lease("tui:voice-tts") + assert calls == ["warm", "warm", "release"] + finally: + tts_registry._reset_for_tests() + + +def test_command_provider_runs_warm_and_release_commands(monkeypatch): + ran: list = [] + done = threading.Event() + + def _fake_run(command, timeout, env_passthrough=None): + ran.append(command) + done.set() + + monkeypatch.setattr(tts_tool, "_run_command_tts", _fake_run) + cfg = { + "provider": "srv", + "providers": {"srv": { + "command": "srv say {input_path} {output_path}", + "warm_command": "curl -s localhost:5002/load?model={model}", + "release_command": "curl -s localhost:5002/unload", + "model": "kokoro v1", + }}, + } + monkeypatch.setattr(tts_tool, "_load_tts_config", lambda: cfg) + + assert tts_tool.acquire_tts_lease("desktop:read-aloud", cfg)["action"] == "warmed" + assert done.wait(5) + done.clear() + tts_tool.release_tts_lease("desktop:read-aloud") + assert done.wait(5) + assert ran == ["curl -s localhost:5002/load?model='kokoro v1'", "curl -s localhost:5002/unload"] diff --git a/tools/tts_tool.py b/tools/tts_tool.py index 60d182e5f9..6b5478a76f 100644 --- a/tools/tts_tool.py +++ b/tools/tts_tool.py @@ -2944,6 +2944,52 @@ _tts_lease_lock = threading.Lock() _tts_leases: set = set() +def _signal_user_tts_provider(name: str, tts_config: Dict[str, Any], hook: str) -> Optional[str]: + """Forward a lease ``hook`` (``"warm"`` / ``"release"``) to a user-declared provider. + + Command providers run their optional ``warm_command`` / ``release_command`` + (same template/env/timeout rules as ``command``; output discarded) on a + background thread so a toggle never waits on a model server. Plugin + providers get :meth:`TTSProvider.warm` / :meth:`TTSProvider.release`. + Best-effort: failures are logged at debug. Returns the action taken. + """ + if not name or name in BUILTIN_TTS_PROVIDERS: + return None + cfg = _get_named_provider_config(tts_config, name) + try: + if _is_command_provider_config(cfg): + template = str(cfg.get(f"{hook}_command") or "").strip() + if not template: + return None + command = _render_command_tts_template(template, { + "voice": str(cfg.get("voice", "")), + "model": str(cfg.get("model", "")), + "speed": str(cfg.get("speed", tts_config.get("speed", ""))), + }) + + def _run() -> None: + try: + _run_command_tts(command, _get_command_tts_timeout(cfg), + env_passthrough=_command_provider_env_passthrough(cfg)) + except Exception as exc: # noqa: BLE001 — best-effort hook + logger.debug("[TTS] %s_command for %s failed: %s", hook, name, exc) + + threading.Thread(target=_run, name=f"tts-{hook}-{name}", daemon=True).start() + return hook + from agent.tts_registry import get_provider + from hermes_cli.plugins import _ensure_plugins_discovered + + _ensure_plugins_discovered() + plugin_provider = get_provider(name) + if plugin_provider is None: + return None + getattr(plugin_provider, hook)() + return hook + except Exception as exc: # noqa: BLE001 — best-effort hook + logger.debug("[TTS] %s hook for %s failed: %s", hook, name, exc) + return "error" + + def warm_tts_provider( tts_config: Optional[Dict[str, Any]] = None, provider: Optional[str] = None, @@ -2955,6 +3001,8 @@ def warm_tts_provider( load it into the same LRU cache slot synthesis reads. * Lazily-installed cloud SDKs (edge-tts, ElevenLabs, Mistral): make sure the SDK is importable, installing it if lazy installs are allowed. + * User-declared providers: command providers run ``warm_command`` when + set; plugin providers get :meth:`TTSProvider.warm`. * Everything else: nothing to warm — reported as ``action: "noop"``. Never raises; the result dict carries ``warmed`` / ``action`` / ``error`` @@ -2986,6 +3034,11 @@ def warm_tts_provider( logger.info("[TTS] warm-up %s: %s in %dms", name, result["action"], result["elapsed_ms"]) return result + signalled = _signal_user_tts_provider(name, tts_config, "warm") + if signalled is not None: + result.update(warmed=signalled != "error", action="warmed" if signalled != "error" else "error") + return result + feature = _lazy_sdk_feature_for_provider(name) if feature is not None: try: @@ -3006,11 +3059,16 @@ def release_tts_provider(provider: Optional[str] = None) -> Dict[str, Any]: """Drop resident local TTS models so their memory is returned. With ``provider`` given, only that engine's cache is cleared; otherwise - every local engine cache is. Cloud providers hold nothing to release. + every local engine cache is and the configured user-declared provider + (plugin ``release()`` / command ``release_command``) is signalled. + Cloud providers hold nothing to release. Returns ``{"released": }``. The next synthesis simply reloads (or a warm-up does it ahead of time). """ name = (provider or "").lower().strip() + if not name: + tts_config = _load_tts_config() + _signal_user_tts_provider(_get_provider(tts_config), tts_config, "release") released = 0 for cache_name, cache in _LOCAL_TTS_MODEL_CACHES.items(): if name and cache_name != name: diff --git a/website/docs/user-guide/features/tts.md b/website/docs/user-guide/features/tts.md index aacc31e3bf..e2ae021a81 100644 --- a/website/docs/user-guide/features/tts.md +++ b/website/docs/user-guide/features/tts.md @@ -267,6 +267,8 @@ Each toggle holds a *lease* on the engine; the model is only unloaded when the l The Desktop calls `POST /api/audio/tts-lease` with `{"lease": "", "active": true|false}`; other frontends can use the same endpoint. +The same lease also reaches user-declared providers, so a self-hosted TTS server can preload and unload its model on the toggles: a [command provider](#custom-command-providers) runs its optional `warm_command` / `release_command`, and a [Python plugin provider](#python-plugin-providers) gets `warm()` / `release()`. + ### Custom command providers If a TTS engine you want isn't natively supported (VoxCPM, MLX-Kokoro, XTTS CLI, a voice-cloning script, anything else that exposes a CLI), you can wire it in as a **command-type provider** without writing any Python. Hermes writes the input text to a temp UTF-8 file, runs your shell command, and reads the audio file the command produced. @@ -359,6 +361,7 @@ Use `{{` and `}}` for literal braces. | `voice_compatible` | `false` | When `true`, Hermes converts MP3/WAV output to Opus/OGG via ffmpeg so Telegram renders a voice bubble. | | `max_text_length` | `5000` | Maximum input characters per command invocation; longer text is split into ordered chunks. | | `voice` / `model` | empty | Passed to the command as placeholder values only. | +| `warm_command` / `release_command` | unset | Shell commands run when a surface toggles speech output on / when the last lease across surfaces is released — e.g. `curl -s localhost:5002/load?model={model}` to preload a local TTS server, and its `unload` counterpart. Best-effort and non-blocking: run in the background with the same `timeout`, `env_passthrough` and `{voice}` / `{model}` / `{speed}` placeholders as `command`; output is discarded and failures are only logged at debug. | #### Behavior notes @@ -448,6 +451,7 @@ Override these on your provider class for richer integration: - `get_setup_schema()` → return `{name, badge, tag, env_vars: [{key, prompt, url}]}` to power the picker row in `hermes tools` / `hermes setup`. Without this, the plugin still works but its row in the picker is minimal. - `stream(text, *, voice, model, format, **extra)` → iterator yielding audio bytes for streaming delivery (default raises `NotImplementedError`). - `voice_compatible` property → set `True` if your output is Opus-compatible and the gateway should deliver it as a voice bubble (default `False` = regular audio attachment). +- `warm()` / `release()` → called when a surface toggles speech output on / when the last lease across surfaces is released, while your provider is the configured `tts.provider` — preload or unload a local model server here. Both default to no-ops; exceptions are logged at debug and never fail the toggle. See `agent/tts_provider.py` for the full ABC including docstrings.