feat(tts): speech toggles warm/release plugin and command TTS providers

Extends the TTS lease from #100912 (4e3feb8bbb) beyond built-in local
engines: when the configured tts.provider is user-declared, acquiring
the first lease and releasing the last one now reach it, so a
self-hosted TTS server can preload its model when read-aloud / voice
conversation turns on and unload when it turns off (Discord request).

- agent/tts_provider.py: TTSProvider gains concrete no-op warm() /
  release() (not abstract — existing plugins are unaffected).
- tools/tts_tool.py: _signal_user_tts_provider() forwards the lease
  hook; plugin providers get warm()/release(), command providers run
  optional `warm_command` / `release_command` (config.yaml, under
  tts.providers.<name>) through the existing _run_command_tts helper
  on a daemon thread — best-effort, output discarded, failures at
  debug. warm_tts_provider() and release_tts_provider() call it.
- tests/tools/test_tts_lifecycle_leases.py: fake plugin provider and
  fake command provider observe warm/release through acquire/release
  lease (both fail on main with action == "noop").
- docs: features/tts.md — lease section, command-provider optional
  keys table, plugin optional hooks.
This commit is contained in:
Teknium
2026-09-02 04:37:13 -07:00
parent 552159d222
commit 782dd635fe
4 changed files with 151 additions and 1 deletions
+16
View File
@@ -241,6 +241,22 @@ class TTSProvider(abc.ABC):
"if your backend supports it."
)
def warm(self) -> None:
"""Speech output was just turned on; pre-load so the first reply is hot.
Optional. Called from the TTS lease path (Desktop read-aloud / voice
conversation, ``/voice tts``) when this provider is the configured
``tts.provider`` — e.g. ask a local model server to load its model.
Best-effort: exceptions are logged at debug and ignored. Default: no-op.
"""
def release(self) -> None:
"""The last speech-output lease was released; free resident resources.
Optional counterpart of :meth:`warm` — e.g. tell a local model server
to unload. Best-effort; default: no-op.
"""
@property
def voice_compatible(self) -> bool:
"""Whether output is suitable for voice-bubble delivery.
+72
View File
@@ -9,6 +9,8 @@ resident local models.
from __future__ import annotations
import threading
import pytest
from tools import tts_tool
@@ -231,3 +233,73 @@ def test_every_local_warmer_has_a_registered_cache():
assert set(warmers) == set(tts_tool._LOCAL_TTS_MODEL_CACHES)
assert tts_tool._LOCAL_TTS_MODEL_CACHES["piper"] is tts_tool._piper_voice_cache
assert tts_tool._LOCAL_TTS_MODEL_CACHES["kittentts"] is tts_tool._kittentts_model_cache
# --------------------------------------------------------------------------
# User-declared providers get the same signal (plugin warm()/release(),
# command warm_command/release_command) so a local TTS server can preload
# and unload on the speech toggles.
# --------------------------------------------------------------------------
def test_plugin_provider_warm_and_release_follow_the_lease(monkeypatch):
from agent import tts_provider, tts_registry
calls: list = []
class _ServerBacked(tts_provider.TTSProvider):
@property
def name(self):
return "my-server"
def synthesize(self, text, output_path, **kw):
return output_path
def warm(self):
calls.append("warm")
def release(self):
calls.append("release")
tts_registry._reset_for_tests()
tts_registry.register_provider(_ServerBacked())
cfg = {"provider": "my-server"}
monkeypatch.setattr(tts_tool, "_load_tts_config", lambda: cfg)
monkeypatch.setattr("hermes_cli.plugins._ensure_plugins_discovered", lambda force=False: None)
try:
assert tts_tool.acquire_tts_lease("desktop:read-aloud", cfg)["action"] == "warmed"
tts_tool.acquire_tts_lease("tui:voice-tts", cfg)
tts_tool.release_tts_lease("desktop:read-aloud")
assert calls == ["warm", "warm"] # still one holder — no release yet
tts_tool.release_tts_lease("tui:voice-tts")
assert calls == ["warm", "warm", "release"]
finally:
tts_registry._reset_for_tests()
def test_command_provider_runs_warm_and_release_commands(monkeypatch):
ran: list = []
done = threading.Event()
def _fake_run(command, timeout, env_passthrough=None):
ran.append(command)
done.set()
monkeypatch.setattr(tts_tool, "_run_command_tts", _fake_run)
cfg = {
"provider": "srv",
"providers": {"srv": {
"command": "srv say {input_path} {output_path}",
"warm_command": "curl -s localhost:5002/load?model={model}",
"release_command": "curl -s localhost:5002/unload",
"model": "kokoro v1",
}},
}
monkeypatch.setattr(tts_tool, "_load_tts_config", lambda: cfg)
assert tts_tool.acquire_tts_lease("desktop:read-aloud", cfg)["action"] == "warmed"
assert done.wait(5)
done.clear()
tts_tool.release_tts_lease("desktop:read-aloud")
assert done.wait(5)
assert ran == ["curl -s localhost:5002/load?model='kokoro v1'", "curl -s localhost:5002/unload"]
+59 -1
View File
@@ -2944,6 +2944,52 @@ _tts_lease_lock = threading.Lock()
_tts_leases: set = set()
def _signal_user_tts_provider(name: str, tts_config: Dict[str, Any], hook: str) -> Optional[str]:
"""Forward a lease ``hook`` (``"warm"`` / ``"release"``) to a user-declared provider.
Command providers run their optional ``warm_command`` / ``release_command``
(same template/env/timeout rules as ``command``; output discarded) on a
background thread so a toggle never waits on a model server. Plugin
providers get :meth:`TTSProvider.warm` / :meth:`TTSProvider.release`.
Best-effort: failures are logged at debug. Returns the action taken.
"""
if not name or name in BUILTIN_TTS_PROVIDERS:
return None
cfg = _get_named_provider_config(tts_config, name)
try:
if _is_command_provider_config(cfg):
template = str(cfg.get(f"{hook}_command") or "").strip()
if not template:
return None
command = _render_command_tts_template(template, {
"voice": str(cfg.get("voice", "")),
"model": str(cfg.get("model", "")),
"speed": str(cfg.get("speed", tts_config.get("speed", ""))),
})
def _run() -> None:
try:
_run_command_tts(command, _get_command_tts_timeout(cfg),
env_passthrough=_command_provider_env_passthrough(cfg))
except Exception as exc: # noqa: BLE001 — best-effort hook
logger.debug("[TTS] %s_command for %s failed: %s", hook, name, exc)
threading.Thread(target=_run, name=f"tts-{hook}-{name}", daemon=True).start()
return hook
from agent.tts_registry import get_provider
from hermes_cli.plugins import _ensure_plugins_discovered
_ensure_plugins_discovered()
plugin_provider = get_provider(name)
if plugin_provider is None:
return None
getattr(plugin_provider, hook)()
return hook
except Exception as exc: # noqa: BLE001 — best-effort hook
logger.debug("[TTS] %s hook for %s failed: %s", hook, name, exc)
return "error"
def warm_tts_provider(
tts_config: Optional[Dict[str, Any]] = None,
provider: Optional[str] = None,
@@ -2955,6 +3001,8 @@ def warm_tts_provider(
load it into the same LRU cache slot synthesis reads.
* Lazily-installed cloud SDKs (edge-tts, ElevenLabs, Mistral): make sure
the SDK is importable, installing it if lazy installs are allowed.
* User-declared providers: command providers run ``warm_command`` when
set; plugin providers get :meth:`TTSProvider.warm`.
* Everything else: nothing to warm — reported as ``action: "noop"``.
Never raises; the result dict carries ``warmed`` / ``action`` / ``error``
@@ -2986,6 +3034,11 @@ def warm_tts_provider(
logger.info("[TTS] warm-up %s: %s in %dms", name, result["action"], result["elapsed_ms"])
return result
signalled = _signal_user_tts_provider(name, tts_config, "warm")
if signalled is not None:
result.update(warmed=signalled != "error", action="warmed" if signalled != "error" else "error")
return result
feature = _lazy_sdk_feature_for_provider(name)
if feature is not None:
try:
@@ -3006,11 +3059,16 @@ def release_tts_provider(provider: Optional[str] = None) -> Dict[str, Any]:
"""Drop resident local TTS models so their memory is returned.
With ``provider`` given, only that engine's cache is cleared; otherwise
every local engine cache is. Cloud providers hold nothing to release.
every local engine cache is and the configured user-declared provider
(plugin ``release()`` / command ``release_command``) is signalled.
Cloud providers hold nothing to release.
Returns ``{"released": <number of model instances dropped>}``. The next
synthesis simply reloads (or a warm-up does it ahead of time).
"""
name = (provider or "").lower().strip()
if not name:
tts_config = _load_tts_config()
_signal_user_tts_provider(_get_provider(tts_config), tts_config, "release")
released = 0
for cache_name, cache in _LOCAL_TTS_MODEL_CACHES.items():
if name and cache_name != name:
+4
View File
@@ -267,6 +267,8 @@ Each toggle holds a *lease* on the engine; the model is only unloaded when the l
The Desktop calls `POST /api/audio/tts-lease` with `{"lease": "<name>", "active": true|false}`; other frontends can use the same endpoint.
The same lease also reaches user-declared providers, so a self-hosted TTS server can preload and unload its model on the toggles: a [command provider](#custom-command-providers) runs its optional `warm_command` / `release_command`, and a [Python plugin provider](#python-plugin-providers) gets `warm()` / `release()`.
### Custom command providers
If a TTS engine you want isn't natively supported (VoxCPM, MLX-Kokoro, XTTS CLI, a voice-cloning script, anything else that exposes a CLI), you can wire it in as a **command-type provider** without writing any Python. Hermes writes the input text to a temp UTF-8 file, runs your shell command, and reads the audio file the command produced.
@@ -359,6 +361,7 @@ Use `{{` and `}}` for literal braces.
| `voice_compatible` | `false` | When `true`, Hermes converts MP3/WAV output to Opus/OGG via ffmpeg so Telegram renders a voice bubble. |
| `max_text_length` | `5000` | Maximum input characters per command invocation; longer text is split into ordered chunks. |
| `voice` / `model` | empty | Passed to the command as placeholder values only. |
| `warm_command` / `release_command` | unset | Shell commands run when a surface toggles speech output on / when the last lease across surfaces is released — e.g. `curl -s localhost:5002/load?model={model}` to preload a local TTS server, and its `unload` counterpart. Best-effort and non-blocking: run in the background with the same `timeout`, `env_passthrough` and `{voice}` / `{model}` / `{speed}` placeholders as `command`; output is discarded and failures are only logged at debug. |
#### Behavior notes
@@ -448,6 +451,7 @@ Override these on your provider class for richer integration:
- `get_setup_schema()` → return `{name, badge, tag, env_vars: [{key, prompt, url}]}` to power the picker row in `hermes tools` / `hermes setup`. Without this, the plugin still works but its row in the picker is minimal.
- `stream(text, *, voice, model, format, **extra)` → iterator yielding audio bytes for streaming delivery (default raises `NotImplementedError`).
- `voice_compatible` property → set `True` if your output is Opus-compatible and the gateway should deliver it as a voice bubble (default `False` = regular audio attachment).
- `warm()` / `release()` → called when a surface toggles speech output on / when the last lease across surfaces is released, while your provider is the configured `tts.provider` — preload or unload a local model server here. Both default to no-ops; exceptions are logged at debug and never fail the toggle.
See `agent/tts_provider.py` for the full ABC including docstrings.