diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index e55b3d923d..c1dd275401 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1667,7 +1667,11 @@ DEFAULT_CONFIG = { # the raw transcript is also echoed back to the user as a 🎙️ message. # Set false to keep STT for the agent while suppressing that user-facing echo. "echo_transcripts": True, - "provider": "local", # "local" (free, faster-whisper) | "groq" | "openai" (Whisper API) | "mistral" (Voxtral Transcribe) | "elevenlabs" (Scribe) | "deepinfra" + # NOTE: no seeded "provider" key. Strict selection semantics treat a + # stored stt.provider as an explicit user pick; seeding "local" here + # made a fresh install indistinguishable from a user choice. The + # autodetect ladder covers unset. Valid values when set: + # "local" (free, faster-whisper) | "groq" | "openai" (Whisper API) | "mistral" (Voxtral Transcribe) | "elevenlabs" (Scribe) | "deepinfra" # Global language hint applied to EVERY provider unless a per-provider # language overrides it. Defaults to "en" — Whisper auto-detection # frequently misidentifies short/accented clips, which reads as diff --git a/tools/transcription_tools.py b/tools/transcription_tools.py index c02a35a934..77f4cca296 100644 --- a/tools/transcription_tools.py +++ b/tools/transcription_tools.py @@ -1025,6 +1025,27 @@ def _get_provider(stt_config: dict) -> str: explicit = "provider" in stt_config provider = stt_config.get("provider", DEFAULT_PROVIDER) + # The managed "Nous Subscription" selection (stt.provider: nous) is + # serviced by the OpenAI provider implementation, routed through the + # managed openai-audio gateway by _resolve_openai_audio_client_config. + if isinstance(provider, str) and provider.strip().lower() == "nous": + provider = "openai" + + if explicit and provider == "local": + # Legacy DEFAULT_CONFIG seeded ``stt.provider: local`` on every + # install, so a merged-config "local" is not proof of a user pick. + # ``read_selection`` reads the raw config.yaml and applies the + # seeded-value migration shim; when the raw file holds no stt + # selection, take the autodetect branch (which prefers local first + # anyway, so a genuine local user is unaffected when it's available). + try: + from tools.tool_backend_helpers import read_selection + + if read_selection("stt") is None: + explicit = False + except Exception: # pragma: no cover — helpers are in-repo + pass + # --- Explicit provider: respect the user's choice ---------------------- if explicit: @@ -3260,11 +3281,61 @@ def transcribe_audio_local_fallback( def _resolve_openai_audio_client_config() -> tuple[str, str]: - """Return direct OpenAI audio config or a managed gateway fallback.""" + """Return ``(api_key, base_url)`` for the OpenAI STT client. + + Strict selection semantics (switch on the stored ``stt`` provider + string; previously this resolver never read the stored gateway intent): + - ``"nous"`` (or legacy ``use_gateway: true``) → managed gateway ONLY; + unentitled/unreachable is a selection-naming error (a direct + OPENAI_API_KEY must NOT override it). + - any other stored stt provider → direct credentials ONLY; missing + credentials is a selection-naming error — no silent managed fallback. + - never-configured stt section → legacy ladder: config key → local + base_url → env key → managed gateway. + """ + from tools.tool_backend_helpers import ( + NOUS_MANAGED_PROVIDER, + read_selection, + selection_error, + ) + stt_config = _load_stt_config() openai_cfg = stt_config.get("openai") or {} cfg_api_key = openai_cfg.get("api_key", "") cfg_base_url = openai_cfg.get("base_url", "") + + selected = read_selection("stt") + + if selected == NOUS_MANAGED_PROVIDER: + managed_gateway = resolve_managed_tool_gateway("openai-audio") + if managed_gateway is None: + raise ValueError(selection_error( + "stt", + NOUS_MANAGED_PROVIDER, + "the Nous Tool Gateway is not available (not entitled or " + "unreachable)", + )) + return managed_gateway.nous_user_token, urljoin( + f"{managed_gateway.gateway_origin.rstrip('/')}/", "v1" + ) + + if selected is not None: + # Stored vendor selection: direct credentials only. + if cfg_api_key: + return cfg_api_key, (cfg_base_url or OPENAI_BASE_URL) + if cfg_base_url and _is_local_or_private_url(cfg_base_url): + return "not-needed", cfg_base_url + direct_api_key = resolve_openai_audio_api_key() + if direct_api_key: + return direct_api_key, OPENAI_BASE_URL + raise ValueError(selection_error( + "stt", + selected, + "neither stt.openai.api_key in config nor " + "VOICE_TOOLS_OPENAI_KEY/OPENAI_API_KEY is set", + )) + + # Never-configured stt section: legacy credential ladder. if cfg_api_key: return cfg_api_key, (cfg_base_url or OPENAI_BASE_URL) diff --git a/tools/tts_tool.py b/tools/tts_tool.py index fba661fd5b..3aa4875922 100644 --- a/tools/tts_tool.py +++ b/tools/tts_tool.py @@ -93,10 +93,12 @@ def _resolve_provider_key(env_var: str, provider_id: str) -> str: from tools.managed_tool_gateway import resolve_managed_tool_gateway from tools.tool_backend_helpers import ( + NOUS_MANAGED_PROVIDER, managed_nous_tools_enabled, nous_tool_gateway_unavailable_message, - prefers_gateway, + read_selection, resolve_openai_audio_api_key, + selection_error, ) from tools.xai_http import hermes_xai_user_agent @@ -651,8 +653,15 @@ def _get_provider(tts_config: Dict[str, Any]) -> str: Inference credentials do not imply consent to paid speech generation. Users opt into cloud TTS by setting ``tts.provider`` (normally through ``hermes tools``); otherwise the historical Edge backend remains active. + + The managed "Nous Subscription" selection (``tts.provider: nous``) is + serviced by the OpenAI provider implementation, routed through the + managed openai-audio gateway by ``_resolve_openai_audio_client_config``. """ - return (tts_config.get("provider") or DEFAULT_PROVIDER).lower().strip() + provider = (tts_config.get("provider") or DEFAULT_PROVIDER).lower().strip() + if provider == NOUS_MANAGED_PROVIDER: + return "openai" + return provider @dataclass(frozen=True) @@ -3785,24 +3794,61 @@ def _resolve_openai_audio_client_config() -> tuple[str, str, bool]: ``is_managed`` is True when the config resolves to the Nous managed audio gateway (a restricted proxy), so callers can coerce the request to what the - gateway supports. When ``tts.use_gateway`` is set the gateway is preferred - even if direct OpenAI credentials are present. + gateway supports. - Resolution order (mirrors the STT resolver): - 1. ``tts.openai.api_key`` / ``tts.openai.base_url`` from ``config.yaml`` - 2. ``VOICE_TOOLS_OPENAI_KEY`` / ``OPENAI_API_KEY`` environment variables - (still honoring ``tts.openai.base_url`` when set) - 3. Managed OpenAI audio tool gateway + Strict selection semantics (switch on the stored ``tts`` provider + string): + - ``"nous"`` (or legacy ``use_gateway: true``) → managed gateway ONLY; + unentitled/unreachable is a selection-naming error. + - any other stored tts provider → direct credentials ONLY + (``tts.openai.api_key`` then ``VOICE_TOOLS_OPENAI_KEY``/ + ``OPENAI_API_KEY``); missing credentials is a selection-naming error — + no silent managed fallback. + - never-configured tts section → legacy ladder: config key → env key → + managed gateway. """ tts_config = _load_tts_config() openai_cfg = (tts_config.get("openai") if isinstance(tts_config, dict) else None) or {} cfg_api_key = openai_cfg.get("api_key") or "" cfg_base_url = openai_cfg.get("base_url") or "" - if cfg_api_key and not prefers_gateway("tts"): + + selected = read_selection("tts") + + if selected == NOUS_MANAGED_PROVIDER: + managed_gateway = resolve_managed_tool_gateway("openai-audio") + if managed_gateway is None: + raise ValueError(selection_error( + "tts", + NOUS_MANAGED_PROVIDER, + "the Nous Tool Gateway is not available (not entitled or " + "unreachable)", + )) + return ( + managed_gateway.nous_user_token, + urljoin(f"{managed_gateway.gateway_origin.rstrip('/')}/", "v1"), + True, + ) + + if selected is not None: + # Stored vendor selection: direct credentials only. + if cfg_api_key: + return cfg_api_key, (cfg_base_url or DEFAULT_OPENAI_BASE_URL), False + direct_api_key = resolve_openai_audio_api_key() + if direct_api_key: + return direct_api_key, (cfg_base_url or DEFAULT_OPENAI_BASE_URL), False + raise ValueError(selection_error( + "tts", + selected, + "neither tts.openai.api_key in config nor " + "VOICE_TOOLS_OPENAI_KEY/OPENAI_API_KEY is set", + )) + + # Never-configured tts section: legacy credential ladder. + if cfg_api_key: return cfg_api_key, (cfg_base_url or DEFAULT_OPENAI_BASE_URL), False direct_api_key = resolve_openai_audio_api_key() - if direct_api_key and not prefers_gateway("tts"): + if direct_api_key: return direct_api_key, (cfg_base_url or DEFAULT_OPENAI_BASE_URL), False managed_gateway = resolve_managed_tool_gateway("openai-audio") @@ -3811,7 +3857,7 @@ def _resolve_openai_audio_client_config() -> tuple[str, str, bool]: "Neither tts.openai.api_key in config nor " "VOICE_TOOLS_OPENAI_KEY/OPENAI_API_KEY is set" ) - if managed_nous_tools_enabled() or prefers_gateway("tts"): + if managed_nous_tools_enabled(): message += ( ". " + nous_tool_gateway_unavailable_message( @@ -3828,11 +3874,12 @@ def _resolve_openai_audio_client_config() -> tuple[str, str, bool]: def _has_openai_audio_backend() -> bool: - """Return True when OpenAI audio can use config/env credentials or the managed gateway.""" - openai_cfg = (_load_tts_config().get("openai") or {}) - if openai_cfg.get("api_key"): + """Return True when the selected OpenAI audio route is usable.""" + try: + _resolve_openai_audio_client_config() return True - return bool(resolve_openai_audio_api_key() or resolve_managed_tool_gateway("openai-audio")) + except ValueError: + return False # ===========================================================================