diff --git a/tools/transcription_audio.py b/tools/transcription_audio.py index 51935a2b80..295b2cfa8c 100644 --- a/tools/transcription_audio.py +++ b/tools/transcription_audio.py @@ -21,8 +21,7 @@ from hermes_cli._subprocess_compat import windows_hide_flags from utils import is_truthy_value from tools.transcription_common import ( COMMON_LOCAL_BIN_DIRS, LOCAL_NATIVE_AUDIO_FORMATS, MAX_FILE_SIZE, SUPPORTED_FORMATS, - _config_number, _error_result, _lazy_ensure_quietly, _process_error_detail, -) + _config_number, _error_result, _lazy_ensure_quietly, _process_error_detail) # Log-record parity with the origin module. logger = logging.getLogger("tools.transcription_tools") @@ -54,8 +53,7 @@ def _run_quiet(command: list, *, timeout: float, env: Optional[dict] = None) -> return subprocess.run( command, check=True, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, - stdin=subprocess.DEVNULL, env=env, creationflags=windows_hide_flags(), - ) + stdin=subprocess.DEVNULL, env=env, creationflags=windows_hide_flags()) # Shared encode profile for every STT-bound m4a (transcode and silence-trim): @@ -268,8 +266,7 @@ def _trim_silence_for_cloud_stt(file_path: str, stt_config: Dict[str, Any]) -> O filter_expr = ( f"silenceremove=" f"start_periods=1:start_threshold={threshold_db}dB:start_silence={keep_seconds}:" - f"stop_periods=-1:stop_threshold={threshold_db}dB:stop_silence={keep_seconds}" - ) + f"stop_periods=-1:stop_threshold={threshold_db}dB:stop_silence={keep_seconds}") work_dir = tempfile.mkdtemp(prefix="hermes-stt-trim-") trimmed_path = os.path.join(work_dir, f"{Path(file_path).stem or 'audio'}-trimmed.m4a") # Scale the all-silence guard with keep_ms: output that is solely kept pause must never upload as "speech". diff --git a/tools/transcription_cloud.py b/tools/transcription_cloud.py index 25714431d8..5627b8a410 100644 --- a/tools/transcription_cloud.py +++ b/tools/transcription_cloud.py @@ -21,8 +21,7 @@ from tools.transcription_audio import _transcode_audio_for_stt from tools.transcription_common import ( DEFAULT_GROQ_STT_MODEL, DEFAULT_STT_MODEL, ELEVENLABS_STT_BASE_URL, GROQ_BASE_URL, GROQ_MODELS, OPENAI_BASE_URL, OPENAI_MODELS, XAI_STT_BASE_URL, _error_result, _get_stt_section, - _lazy_ensure_quietly, _log_prompt_unsupported, _ok_result, -) + _lazy_ensure_quietly, _log_prompt_unsupported, _ok_result) # Log-record parity with the origin module. logger = logging.getLogger("tools.transcription_tools") @@ -110,8 +109,7 @@ def _transcribe_groq( def _transcribe_openai( file_path: str, model_name: str, *, api_key: Optional[str] = None, base_url: Optional[str] = None, provider_label: str = "openai", language: Optional[str] = None, - prompt: Optional[str] = None, -) -> Dict[str, Any]: + prompt: Optional[str] = None) -> Dict[str, Any]: """Transcribe via the OpenAI ``audio.transcriptions.create`` SDK shape. Shared by every OpenAI-compatible endpoint (DeepInfra etc.): explicit ``api_key``/ @@ -189,8 +187,7 @@ def _transcribe_mistral( language = language or _resolve_stt_language("mistral") result = client.audio.transcriptions.complete( model=model_name, file={"content": audio_file, "file_name": Path(file_path).name}, - **_sdk_prompt_kwargs(language, prompt), - ) + **_sdk_prompt_kwargs(language, prompt)) transcript_text = _extract_transcript_text(result) logger.info("Transcribed %s via Mistral API (%s, %d chars)", Path(file_path).name, model_name, len(transcript_text)) @@ -210,8 +207,7 @@ def _post_audio_multipart(url: str, headers: Dict[str, str], file_path: str, dat def _rest_provider( file_path: str, provider: str, label: str, post: Callable[[], Any], extract_detail, - extract_text, log: Callable[[str, Dict[str, Any]], None], -) -> Dict[str, Any]: + extract_text, log: Callable[[str, Dict[str, Any]], None]) -> Dict[str, Any]: """Shared multipart REST flow: ``post()`` -> ``log(text, body)`` -> ok envelope. Non-200 -> ``"