feat(stt): add language parameter support for OpenAI provider
Add optional stt.openai.language config for OpenAI transcription. Forward non-empty hints to the API while preserving auto-detection when unset. Document the config-only setting, add its default, and cover configured and unset request arguments.
This commit is contained in:
@@ -1129,6 +1129,7 @@ stt:
|
||||
# language: "" # auto-detect; set to "en", "es", "fr", etc. to force
|
||||
openai:
|
||||
model: "whisper-1" # whisper-1 | gpt-4o-mini-transcribe | gpt-4o-transcribe
|
||||
language: "" # auto-detect; set to "en", "es", "fr", etc. to force
|
||||
# mistral:
|
||||
# model: "voxtral-mini-latest" # voxtral-mini-latest | voxtral-mini-2602
|
||||
# deepinfra:
|
||||
|
||||
@@ -2328,6 +2328,7 @@ DEFAULT_CONFIG = {
|
||||
},
|
||||
"openai": {
|
||||
"model": "whisper-1", # whisper-1, gpt-4o-mini-transcribe, gpt-4o-transcribe
|
||||
"language": "", # auto-detect by default; set to "en", "es", "fr", etc. to force
|
||||
},
|
||||
"mistral": {
|
||||
"model": "voxtral-mini-latest", # voxtral-mini-latest, voxtral-mini-2602
|
||||
|
||||
@@ -190,6 +190,44 @@ class TestTranscribeOpenAI:
|
||||
assert result["success"] is True
|
||||
assert result["transcript"] == "Hello from OpenAI"
|
||||
|
||||
def test_configured_language_is_forwarded(self, monkeypatch, tmp_path):
|
||||
monkeypatch.setenv("VOICE_TOOLS_OPENAI_KEY", "sk-test")
|
||||
audio_file = tmp_path / "test.ogg"
|
||||
audio_file.write_bytes(b"fake audio")
|
||||
|
||||
mock_client = MagicMock()
|
||||
mock_client.audio.transcriptions.create.return_value = "Привіт"
|
||||
|
||||
with patch("tools.transcription_tools._HAS_OPENAI", True), \
|
||||
patch("tools.transcription_tools._load_stt_config", return_value={
|
||||
"openai": {"language": "uk"},
|
||||
}), \
|
||||
patch("openai.OpenAI", return_value=mock_client):
|
||||
from tools.transcription_tools import _transcribe_openai
|
||||
result = _transcribe_openai(str(audio_file), "whisper-1")
|
||||
|
||||
assert result["success"] is True
|
||||
assert mock_client.audio.transcriptions.create.call_args.kwargs["language"] == "uk"
|
||||
|
||||
def test_unset_language_omits_argument(self, monkeypatch, tmp_path):
|
||||
monkeypatch.setenv("VOICE_TOOLS_OPENAI_KEY", "sk-test")
|
||||
audio_file = tmp_path / "test.ogg"
|
||||
audio_file.write_bytes(b"fake audio")
|
||||
|
||||
mock_client = MagicMock()
|
||||
mock_client.audio.transcriptions.create.return_value = "Hello"
|
||||
|
||||
with patch("tools.transcription_tools._HAS_OPENAI", True), \
|
||||
patch("tools.transcription_tools._load_stt_config", return_value={
|
||||
"openai": {"language": ""},
|
||||
}), \
|
||||
patch("openai.OpenAI", return_value=mock_client):
|
||||
from tools.transcription_tools import _transcribe_openai
|
||||
result = _transcribe_openai(str(audio_file), "whisper-1")
|
||||
|
||||
assert result["success"] is True
|
||||
assert "language" not in mock_client.audio.transcriptions.create.call_args.kwargs
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Main transcribe_audio() dispatch
|
||||
|
||||
@@ -1376,6 +1376,11 @@ def _transcribe_openai(
|
||||
return {"success": False, "transcript": "", "error": str(exc)}
|
||||
base_url = base_url or fallback_base
|
||||
|
||||
# Language: config.yaml (stt.openai.language) > auto-detect.
|
||||
# Explicit language hint improves accuracy for non-English languages.
|
||||
stt_config = _load_stt_config()
|
||||
language = stt_config.get("openai", {}).get("language")
|
||||
|
||||
if not _HAS_OPENAI:
|
||||
return {"success": False, "transcript": "", "error": "openai package not installed"}
|
||||
|
||||
@@ -1391,11 +1396,16 @@ def _transcribe_openai(
|
||||
client = OpenAI(api_key=api_key, base_url=base_url, timeout=30, max_retries=0)
|
||||
try:
|
||||
with open(file_path, "rb") as audio_file:
|
||||
transcription = client.audio.transcriptions.create(
|
||||
model=model_name,
|
||||
file=audio_file,
|
||||
response_format="text" if model_name == "whisper-1" else "json",
|
||||
)
|
||||
create_kwargs = {
|
||||
"model": model_name,
|
||||
"file": audio_file,
|
||||
"response_format": "text" if model_name == "whisper-1" else "json",
|
||||
}
|
||||
if language:
|
||||
create_kwargs["language"] = language
|
||||
logger.debug("Using language hint '%s' for OpenAI STT", language)
|
||||
|
||||
transcription = client.audio.transcriptions.create(**create_kwargs)
|
||||
|
||||
transcript_text = _extract_transcript_text(transcription)
|
||||
logger.info(
|
||||
|
||||
@@ -1717,6 +1717,7 @@ stt:
|
||||
model: "base" # tiny, base, small, medium, large-v3
|
||||
openai:
|
||||
model: "whisper-1" # whisper-1 | gpt-4o-mini-transcribe | gpt-4o-transcribe
|
||||
language: "" # auto-detect; set to "en", "es", "fr", etc. to force
|
||||
# model: "whisper-1" # Legacy fallback key still respected
|
||||
```
|
||||
|
||||
|
||||
Reference in New Issue
Block a user