"""Cloud STT providers. OpenAI-SDK-shaped backends (groq, openai, deepinfra), Mistral Voxtral, the REST multipart backends (xAI, ElevenLabs), and OpenAI audio credential resolution (config > keyless local server > env > managed Nous gateway). Split out of ``tools/transcription_tools.py``, which re-imports every name (patch surface) and is imported lazily here so origin patches still intercept. """ from __future__ import annotations import logging import re import tempfile from pathlib import Path from typing import Any, Dict, Optional from urllib.parse import urljoin from utils import is_truthy_value from tools.transcription_audio import _transcode_audio_for_stt from tools.transcription_common import ( DEFAULT_GROQ_STT_MODEL, DEFAULT_STT_MODEL, ELEVENLABS_STT_BASE_URL, GROQ_BASE_URL, GROQ_MODELS, OPENAI_BASE_URL, OPENAI_MODELS, XAI_STT_BASE_URL, _error_result, _get_stt_section, _lazy_ensure_quietly, _log_prompt_unsupported, _ok_result, ) # Log-record parity with the origin module. logger = logging.getLogger("tools.transcription_tools") def _has_xai_stt_credentials() -> bool: from tools.xai_http import resolve_xai_http_credentials return bool(resolve_xai_http_credentials().get("api_key")) def _close_client(client: Any) -> None: close = getattr(client, "close", None) if callable(close): close() def _with_openai_client(api_key: str, base_url: Optional[str], file_path: str, log_label: str, body): """Run ``body(client)`` against a fresh OpenAI SDK client (30s timeout, no SDK retries). Always closes the client; any exception maps to the shared envelope via :func:`_openai_sdk_failure`. """ try: from openai import OpenAI client = OpenAI(api_key=api_key, base_url=base_url, timeout=30, max_retries=0) try: return body(client) finally: _close_client(client) except Exception as e: return _openai_sdk_failure(e, file_path, log_label) def _cloud_failure(exc: BaseException, file_path: str, label: str, detail: Optional[str] = None) -> Dict[str, Any]: """Map a REST/SDK provider exception to the shared envelope (``label`` e.g. ``"xAI STT transcription"``).""" if isinstance(exc, PermissionError): return _error_result(f"Permission denied: {file_path}") logger.error("%s failed: %s", label, exc, exc_info=True) return _error_result(f"{label} failed: {exc if detail is None else detail}") def _openai_sdk_failure(exc: BaseException, file_path: str, log_label: str) -> Dict[str, Any]: """Map an OpenAI-SDK-shaped exception to the shared error envelope. Order matters: APIConnectionError is checked before APITimeoutError (its subclass) so timeouts report as connection errors, as they always have. """ try: from openai import APIError, APIConnectionError, APITimeoutError except ImportError: # pragma: no cover — callers gate on _HAS_OPENAI APIError = APIConnectionError = APITimeoutError = () if isinstance(exc, PermissionError): return _error_result(f"Permission denied: {file_path}") if isinstance(exc, APIConnectionError): return _error_result(f"Connection error: {exc}") if isinstance(exc, APITimeoutError): return _error_result(f"Request timeout: {exc}") if isinstance(exc, APIError): return _error_result(f"API error: {exc}") logger.error("%s transcription failed: %s", log_label, exc, exc_info=True) return _error_result(f"Transcription failed: {exc}") def _transcribe_groq( file_path: str, model_name: str, *, language: Optional[str] = None, prompt: Optional[str] = None, ) -> Dict[str, Any]: """Transcribe using Groq Whisper API (free tier available). Language: hook override > ``stt.groq.language`` > ``stt.language`` > env; otherwise Groq auto-detects. """ from tools.transcription_tools import _HAS_OPENAI, _resolve_provider_key, _resolve_stt_language api_key = _resolve_provider_key("GROQ_API_KEY", "groq") if not api_key: return _error_result("GROQ_API_KEY not set") if not _HAS_OPENAI: return _error_result("openai package not installed") # Auto-correct model if caller passed an OpenAI-only model if model_name in OPENAI_MODELS: logger.info("Model %s not available on Groq, using %s", model_name, DEFAULT_GROQ_STT_MODEL) model_name = DEFAULT_GROQ_STT_MODEL language = language or _resolve_stt_language("groq") def _run(client): create_kwargs = { "model": model_name, "response_format": "text", } if language: create_kwargs["language"] = language if prompt: # Only sent when set so the no-hook, no-config request stays byte-identical. create_kwargs["prompt"] = prompt with open(file_path, "rb") as audio_file: transcription = client.audio.transcriptions.create( file=audio_file, **create_kwargs, ) transcript_text = str(transcription).strip() logger.info("Transcribed %s via Groq API (%s, lang=%s, %d chars)", Path(file_path).name, model_name, language or "auto", len(transcript_text)) return _ok_result(transcript_text, "groq") return _with_openai_client(api_key, GROQ_BASE_URL, file_path, "Groq", _run) def _transcribe_openai( file_path: str, model_name: str, *, api_key: Optional[str] = None, base_url: Optional[str] = None, provider_label: str = "openai", language: Optional[str] = None, prompt: Optional[str] = None, ) -> Dict[str, Any]: """Transcribe via the OpenAI ``audio.transcriptions.create`` SDK shape. Shared backend for every OpenAI-compatible STT endpoint (DeepInfra etc.): callers pass explicit ``api_key``/``base_url`` to skip the OpenAI-only auth chain and a ``provider_label`` for the response's ``provider``. """ from tools.transcription_tools import _HAS_OPENAI, _resolve_openai_audio_client_config, _resolve_stt_language if api_key is None: try: api_key, fallback_base = _resolve_openai_audio_client_config() except ValueError as exc: return _error_result(str(exc)) base_url = base_url or fallback_base # Language: hook override > stt..language > stt.language > env > auto. language = language or _resolve_stt_language(provider_label) if not _HAS_OPENAI: return _error_result("openai package not installed") # Auto-correct a Groq-only model on the native OpenAI path only — # third-party endpoints may legitimately serve a whisper-large-v3 variant. if provider_label == "openai" and model_name in GROQ_MODELS: logger.info("Model %s not available on OpenAI, using %s", model_name, DEFAULT_STT_MODEL) model_name = DEFAULT_STT_MODEL def _run(client): from openai import BadRequestError def _create_transcription(path: str): with open(path, "rb") as audio_file: create_kwargs = { "model": model_name, "file": audio_file, "response_format": "text" if model_name == "whisper-1" else "json", } if language: if model_name == "gpt-transcribe": # gpt-transcribe replaces ``language`` with a ``languages`` # list and rejects requests sending the legacy field. create_kwargs["extra_body"] = {"languages": [language]} else: create_kwargs["language"] = language logger.debug("Using language hint '%s' for OpenAI STT", language) if prompt: # Only sent when set so the no-hook, no-config request stays byte-identical. create_kwargs["prompt"] = prompt return client.audio.transcriptions.create(**create_kwargs) with tempfile.TemporaryDirectory(prefix="hermes-stt-") as work_dir: try: transcription = _create_transcription(file_path) except BadRequestError as exc: message = str(exc).lower() if not any(k in message for k in ("unsupported", "corrupted", "invalid file")): raise # Newer models reject some containers whisper-1 accepted # (notably Ogg/Opus voice notes): transcode to m4a, retry once. converted_path, transcode_error = _transcode_audio_for_stt(file_path, work_dir) if transcode_error: return _error_result(transcode_error) logger.info( "Retrying %s STT after transcoding %s to m4a (API rejected the original container)", provider_label, Path(file_path).name, ) transcription = _create_transcription(converted_path) transcript_text = _extract_transcript_text(transcription) logger.info( "Transcribed %s via %s (%s, %d chars)", Path(file_path).name, provider_label, model_name, len(transcript_text), ) return _ok_result(transcript_text, provider_label) return _with_openai_client(api_key, base_url, file_path, provider_label, _run) def _transcribe_mistral( file_path: str, model_name: str, *, language: Optional[str] = None, prompt: Optional[str] = None, ) -> Dict[str, Any]: """Transcribe with the ``mistralai`` SDK (``/v1/audio/transcriptions``); requires ``MISTRAL_API_KEY``.""" from tools.transcription_tools import _resolve_provider_key, _resolve_stt_language api_key = _resolve_provider_key("MISTRAL_API_KEY", "mistral") if not api_key: return _error_result("MISTRAL_API_KEY not set") try: _lazy_ensure_quietly("stt.mistral") from mistralai.client import Mistral with Mistral(api_key=api_key) as client: with open(file_path, "rb") as audio_file: complete_kwargs: Dict[str, Any] = { "model": model_name, "file": {"content": audio_file, "file_name": Path(file_path).name}, } # Language: hook override > stt.mistral.language > stt.language > env > auto. language = language or _resolve_stt_language("mistral") if language: complete_kwargs["language"] = language if prompt: # Only sent when set so the no-hook, no-config request stays byte-identical. complete_kwargs["prompt"] = prompt result = client.audio.transcriptions.complete(**complete_kwargs) transcript_text = _extract_transcript_text(result) logger.info( "Transcribed %s via Mistral API (%s, %d chars)", Path(file_path).name, model_name, len(transcript_text), ) return _ok_result(transcript_text, "mistral") except Exception as e: return _cloud_failure(e, file_path, "Mistral transcription", type(e).__name__) def _post_audio_multipart(url: str, headers: Dict[str, str], file_path: str, data: Dict[str, str]): import requests with open(file_path, "rb") as audio_file: return requests.post( url, headers=headers, files={"file": (Path(file_path).name, audio_file)}, data=data, timeout=120, ) def _http_error_detail(response, extract) -> str: """``extract(json_body)`` -> detail string, falling back to the first 300 chars of the body.""" try: return extract(response.json()) or response.text[:300] except Exception: return response.text[:300] def _rest_transcript(response, label: str, extract_detail, extract_text): """Turn a multipart STT response into ``(text, body, None)`` or ``(None, None, error_envelope)``. Non-200 -> ``"