voice: one macOS output policy across WAV, beeps, streaming TTS + tests

Addresses review on #62601. Applies a single rule — no sounddevice for
audio OUTPUT on macOS (PortAudio/CoreAudio init triggers a
kTCCServiceMediaLibrary prompt) — consistently at all three output sites:

- play_audio_file: WAV playback (already routed to afplay) now uses the
  shared _sounddevice_output_allowed() helper.
- play_beep: synthesize the tone with numpy only, then on macOS play it
  via a temp WAV through afplay instead of sounddevice.
- stream_tts_to_speaker (tts_tool): on macOS, skip the sounddevice
  OutputStream so playback falls through to the existing tempfile/afplay path.

Audio INPUT (recording) is untouched — it legitimately needs mic permission.

Tests: TestMacOSAudioOutputPolicy (voice_mode) proves WAV + beep routing
does not import sounddevice on Darwin and still uses it off Darwin;
test_tts_macos_output proves streaming TTS skips the OutputStream on Darwin.
Existing test_play_wav_via_sounddevice pinned to non-Darwin for determinism.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
simonmmafs
2026-07-11 22:06:15 +02:00
committed by Teknium
parent 0179ff9738
commit a1bc12f191
4 changed files with 294 additions and 19 deletions
+68 -6
View File
@@ -42,6 +42,55 @@ def _import_audio():
return sd, np
def _import_numpy():
"""Lazy-import numpy only (no sounddevice). Returns the module.
Used where we need to synthesize/convert audio samples but must NOT
import sounddevice — see _sounddevice_output_allowed.
"""
import numpy as np
return np
def _sounddevice_output_allowed() -> bool:
"""Whether sounddevice may be used for audio OUTPUT.
Returns False on macOS: importing/initializing sounddevice
(PortAudio/CoreAudio) for output triggers a kTCCServiceMediaLibrary
permission prompt, even though playback needs no media-library access.
On macOS all output is routed through ``afplay`` instead. This does NOT
affect audio *input* (recording), which legitimately needs microphone
permission. See PR #62601 / #13291.
"""
return platform.system() != "Darwin"
def _play_int16_via_tempfile(audio, sample_rate: int) -> None:
"""Write int16 mono PCM to a temp WAV and play it via play_audio_file.
Used on macOS so tone/beep output goes through ``afplay`` instead of
sounddevice (avoids the TCC media-library prompt).
"""
tmp_path = None
try:
tmp = tempfile.NamedTemporaryFile(suffix=".wav", delete=False)
tmp_path = tmp.name
with wave.open(tmp, "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2) # 16-bit
wf.setframerate(sample_rate)
wf.writeframes(audio.tobytes())
play_audio_file(tmp_path)
except Exception as e:
logger.debug("Tone tempfile playback failed: %s", e)
finally:
if tmp_path:
try:
os.unlink(tmp_path)
except OSError:
pass
def _audio_available() -> bool:
"""Return True if audio libraries can be imported."""
try:
@@ -419,9 +468,11 @@ def play_beep(frequency: int = 880, duration: float = 0.12, count: int = 1) -> N
duration: Duration of each beep in seconds.
count: Number of beeps to play (with short gap between).
"""
# Synthesize the tone with numpy only (no sounddevice import yet, so the
# macOS TCC prompt is not triggered on the synthesis step).
try:
sd, np = _import_audio()
except (ImportError, OSError):
np = _import_numpy()
except ImportError:
return
try:
gap = 0.06 # seconds between beeps
@@ -442,6 +493,16 @@ def play_beep(frequency: int = 880, duration: float = 0.12, count: int = 1) -> N
parts.append(np.zeros(samples_per_gap, dtype=np.int16))
audio = np.concatenate(parts)
# On macOS, route the tone through afplay instead of sounddevice.
if not _sounddevice_output_allowed():
_play_int16_via_tempfile(audio, SAMPLE_RATE)
return
try:
sd, _ = _import_audio()
except (ImportError, OSError):
return
sd.play(audio, samplerate=SAMPLE_RATE)
# sd.wait() calls Event.wait() without timeout — hangs forever if the
# audio device stalls. Poll with a 2s ceiling and force-stop.
@@ -1306,10 +1367,11 @@ def play_audio_file(file_path: str) -> bool:
logger.warning("Audio file not found: %s", file_path)
return False
# On macOS, skip sounddevice entirely — PortAudio/CoreAudio init triggers
# a kTCCServiceMediaLibrary permission prompt even though we don't need it.
# afplay handles all formats natively without touching the media stack.
if file_path.endswith(".wav") and platform.system() != "Darwin":
# Skip sounddevice for output where it is not allowed (macOS): PortAudio/
# CoreAudio init triggers a kTCCServiceMediaLibrary permission prompt even
# though playback needs no media-library access. afplay (added to the
# system-player list below) handles all formats natively instead.
if file_path.endswith(".wav") and _sounddevice_output_allowed():
try:
sd, np = _import_audio()
with wave.open(file_path, "rb") as wf: