diff --git a/tools/transcription_audio.py b/tools/transcription_audio.py index f34e052ed8..00d6c25c25 100644 --- a/tools/transcription_audio.py +++ b/tools/transcription_audio.py @@ -58,15 +58,14 @@ _STT_M4A_ENCODE_ARGS = ("-vn", "-ac", "1", "-ar", "16000", "-c:a", "aac", "-b:a" def _run_ffmpeg_stt_encode(ffmpeg: str, input_path: str, output_path: str, *, audio_filter: Optional[str] = None) -> None: """Run the shared STT m4a encode, optionally with an ``-af`` filter. Raises on failure; callers own the semantics.""" filter_args = ["-af", audio_filter] if audio_filter else [] - _run_quiet([ffmpeg, "-y", "-i", input_path, *filter_args, *_STT_M4A_ENCODE_ARGS, output_path], timeout=120) + _run_quiet([ffmpeg, "-y", "-i", input_path, *filter_args, *_STT_M4A_ENCODE_ARGS, output_path], + timeout=120) def _transcode_audio_for_stt(file_path: str, work_dir: str) -> tuple[Optional[str], Optional[str]]: """Transcode to a compact 16 kHz mono AAC/m4a for STT upload; ``(converted_path, None)`` or ``(None, error)``. - - Newer OpenAI models reject containers ``whisper-1`` accepted (notably Ogg/Opus - voice notes) and gateway downloads may carry a misleading extension. - """ + Newer OpenAI models reject containers ``whisper-1`` accepted (notably Ogg/Opus voice notes) and + gateway downloads may carry a misleading extension.""" from tools.transcription_tools import _find_ffmpeg_binary, _run_ffmpeg_stt_encode ffmpeg = _find_ffmpeg_binary() if not ffmpeg: @@ -207,18 +206,17 @@ _CLOUD_TRIM_MIN_INPUT_SECONDS = 12.0 def _probe_audio_duration(file_path: str) -> Optional[float]: - """Return the audio duration in seconds via ffprobe, or None. - - Canonical sync probe; ``gateway/run.py._probe_audio_duration`` and the - Telegram adapter carry local variants — keep the command shape in sync. - """ + """Return the audio duration in seconds via ffprobe, or None. Canonical sync probe; + ``gateway/run.py._probe_audio_duration`` and the Telegram adapter carry local variants — keep + the command shape in sync.""" from tools.transcription_tools import _find_ffprobe_binary ffprobe = _find_ffprobe_binary() if not ffprobe: return None try: - return float(_run_quiet([ffprobe, "-v", "error", "-show_entries", "format=duration", "-of", - "default=noprint_wrappers=1:nokey=1", file_path], timeout=30).stdout.strip()) + probe = _run_quiet([ffprobe, "-v", "error", "-show_entries", "format=duration", "-of", + "default=noprint_wrappers=1:nokey=1", file_path], timeout=30) + return float(probe.stdout.strip()) except Exception: # noqa: BLE001 - probe is best-effort return None @@ -235,9 +233,7 @@ def _cloud_trim_settings(stt_config: Dict[str, Any]) -> tuple[bool, int, int]: def _trim_silence_for_cloud_stt(file_path: str, stt_config: Dict[str, Any]) -> Optional[str]: """Return a silence-trimmed copy of *file_path* for cloud upload, or None (= upload the original). - - On success the caller owns deleting the returned file's parent directory. - """ + On success the caller owns deleting the returned file's parent directory.""" from tools.transcription_tools import _find_ffmpeg_binary, _probe_audio_duration, _run_ffmpeg_stt_encode enabled, threshold_db, keep_ms = _cloud_trim_settings(stt_config) if not enabled: @@ -277,8 +273,8 @@ def _trim_silence_for_cloud_stt(file_path: str, stt_config: Dict[str, Any]) -> O logger.debug("Cloud STT silence trim discarded for %s: saves <%.0f%% (%.1fs -> %.1fs)", name, _CLOUD_TRIM_MIN_SAVING * 100, original_duration, trimmed_duration) return None - logger.info("Trimmed silence from %s before cloud STT upload (%.1fs -> %.1fs, -%d%%)", - name, original_duration, trimmed_duration, round((1 - trimmed_duration / original_duration) * 100)) + logger.info("Trimmed silence from %s before cloud STT upload (%.1fs -> %.1fs, -%d%%)", name, + original_duration, trimmed_duration, round((1 - trimmed_duration / original_duration) * 100)) keep_result = True return trimmed_path except Exception as exc: # noqa: BLE001 - trim is best-effort diff --git a/tools/transcription_cloud.py b/tools/transcription_cloud.py index 5627b8a410..aca2f9a2de 100644 --- a/tools/transcription_cloud.py +++ b/tools/transcription_cloud.py @@ -38,10 +38,8 @@ def _has_xai_stt_credentials() -> bool: def _with_openai_client(api_key: str, base_url: Optional[str], file_path: str, log_label: str, body): """Run ``body(client)`` on a fresh OpenAI SDK client (30s timeout, no retries); always closed. - - Errors map to the shared envelope. APIConnectionError is checked before APITimeoutError - (its subclass) so timeouts report as connection errors, as they always have. - """ + Errors map to the shared envelope. APIConnectionError is checked before APITimeoutError (its + subclass) so timeouts report as connection errors, as they always have.""" try: from openai import OpenAI client = OpenAI(api_key=api_key, base_url=base_url, timeout=30, max_retries=0) @@ -96,13 +94,13 @@ def _transcribe_groq( def _run(client): with open(file_path, "rb") as audio_file: - transcription = client.audio.transcriptions.create( - file=audio_file, model=model_name, response_format="text", **_sdk_prompt_kwargs(language, prompt)) + transcription = client.audio.transcriptions.create(file=audio_file, model=model_name, + response_format="text", + **_sdk_prompt_kwargs(language, prompt)) transcript_text = str(transcription).strip() logger.info("Transcribed %s via Groq API (%s, lang=%s, %d chars)", Path(file_path).name, model_name, language or "auto", len(transcript_text)) return _ok_result(transcript_text, "groq") - return _with_openai_client(api_key, GROQ_BASE_URL, file_path, "Groq", _run) @@ -110,11 +108,9 @@ def _transcribe_openai( file_path: str, model_name: str, *, api_key: Optional[str] = None, base_url: Optional[str] = None, provider_label: str = "openai", language: Optional[str] = None, prompt: Optional[str] = None) -> Dict[str, Any]: - """Transcribe via the OpenAI ``audio.transcriptions.create`` SDK shape. - - Shared by every OpenAI-compatible endpoint (DeepInfra etc.): explicit ``api_key``/ - ``base_url`` skip the OpenAI-only auth chain; ``provider_label`` names the response's provider. - """ + """Transcribe via the OpenAI ``audio.transcriptions.create`` SDK shape, shared by every + OpenAI-compatible endpoint (DeepInfra etc.): explicit ``api_key``/``base_url`` skip the + OpenAI-only auth chain; ``provider_label`` names the response's provider.""" from tools.transcription_tools import _HAS_OPENAI, _resolve_openai_audio_client_config, _resolve_stt_language if api_key is None: try: @@ -149,7 +145,6 @@ def _transcribe_openai( create_kwargs["prompt"] = prompt with open(path, "rb") as audio_file: return client.audio.transcriptions.create(file=audio_file, **create_kwargs) - with tempfile.TemporaryDirectory(prefix="hermes-stt-") as work_dir: try: transcription = _create_transcription(file_path) @@ -167,7 +162,6 @@ def _transcribe_openai( logger.info("Transcribed %s via %s (%s, %d chars)", Path(file_path).name, provider_label, model_name, len(transcript_text)) return _ok_result(transcript_text, provider_label) - return _with_openai_client(api_key, base_url, file_path, provider_label, _run) @@ -197,7 +191,6 @@ def _transcribe_mistral( # ---- REST multipart backends (xAI, ElevenLabs) ---------------------------- - def _post_audio_multipart(url: str, headers: Dict[str, str], file_path: str, data: Dict[str, str]): import requests with open(file_path, "rb") as audio_file: @@ -208,12 +201,10 @@ def _post_audio_multipart(url: str, headers: Dict[str, str], file_path: str, dat def _rest_provider( file_path: str, provider: str, label: str, post: Callable[[], Any], extract_detail, extract_text, log: Callable[[str, Dict[str, Any]], None]) -> Dict[str, Any]: - """Shared multipart REST flow: ``post()`` -> ``log(text, body)`` -> ok envelope. - - Non-200 -> ``"