c5ff900761
Sweep of `ruff check . --select F821 --target-version py311`: 2,234 hits. 2,201 are left
alone on purpose: tui_gateway (2,169; bind_module rebinds bodies onto server.py globals,
all names verified to resolve there), the Feishu adapter (27; globals().update() SDK
binding) and the godmode script (5; dead standalone script). The other 33 were all
genuine defects. No lint config change; no TYPE_CHECKING escape hatches — every
annotation names a real, imported type; ty on the touched files: 0 new diagnostics.
- gateway/slash_commands.py: HISTORY_UNREADABLE never imported after #102117
→ NameError on the /btw error branch (same one-liner as #102952).
- gateway/platforms/whatsapp_common.py: `-> Path` return annotation with no Path
import (the body uses `_Path`). Never raised at runtime thanks to
`from __future__ import annotations`, but `typing.get_type_hints()` and ty
both fail on it.
- gateway/run.py: ActivityProvenance imported at module level
(agent.session_activity has no gateway deps); stringly annotation and the
lazy in-function import are gone.
- tools/patch_parser.py: PatchResult imported at module level; real return
annotation. The "avoid circular import" lazy import guarded a cycle that
does not exist (file_operations_common never imports patch_parser).
- gateway/platforms/helpers.py: base.py imports helpers at module level, so
MessageEvent cannot be named here; TextBatchAggregator only reads .text and
.source, so it is typed by a BatchableEvent Protocol that MessageEvent
satisfies structurally.
- tools/mcp_tool_sampling.py: mcp_tool imports this module, so MCPServerTask
cannot be named here; ElicitationHandler only reads
owner._pending_call_context, typed by an ElicitationOwner Protocol.
- plugins/platforms/sms/adapter.py: aiohttp is an optional dep ([messaging] extra) →
module-level try/except ImportError binding `aiohttp = web = None`, the pattern the
homeassistant / webhook / whatsapp_cloud adapters already use. Retires three lazy
in-function imports and the `_aiohttp_available()` wrapper; `_handle_webhook` typed
`web.Request -> web.Response`.
- plugins/platforms/teams/summary_writer.py: plain module-level `import httpx` — httpx is a
hard core dependency (pyproject `httpx[socks]==0.28.1`), so the lazy import and the
"imported on every CLI start" docstring premise were both wrong (plugin discovery never
imports this module; it is reached only via the Teams adapter / meeting pipeline).
Tests:
- tests/hermes_cli/test_config.py: a test body orphaned by the wave-1 prune
(6b81590c55) sat inside the class as dead code with self/tmp_path unbound
— header restored, so the v11→12 custom_providers migration is covered.
- tests/tools/test_mcp_tool.py: @staticmethod recursing on `self` in the
win32 branch; call portalocker directly.
- tests/test_background_review_list_shapes.py: main() still ran 3 pruned tests.
- tests/agent/test_cursor_optimizations_parity.py: bench() used names only
imported inside a sibling test.
- GatewayRunner / FeishuAdapter / Dict / Optional: missing imports.
133 lines
4.8 KiB
Python
133 lines
4.8 KiB
Python
"""
|
|
Tests for #24870 — Telegram: audio file attachments must NOT be routed to STT.
|
|
|
|
Telegram distinguishes three kinds of audio payloads:
|
|
- message.voice → Opus/OGG voice message → STT pipeline
|
|
- message.audio → audio file attachment → file path note, NOT STT
|
|
- message.document (audio mime) → generic file route
|
|
|
|
These tests confirm that:
|
|
1. MessageType.VOICE events still flow through the STT pipeline.
|
|
2. MessageType.AUDIO events bypass STT and get a file-path context note instead.
|
|
3. Mixed media lists (voice + audio) split correctly.
|
|
"""
|
|
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
from gateway.config import GatewayConfig, Platform
|
|
from gateway.platforms.base import MessageEvent, MessageType
|
|
from gateway.session import SessionSource
|
|
|
|
from typing import TYPE_CHECKING
|
|
|
|
if TYPE_CHECKING:
|
|
from gateway.run import GatewayRunner
|
|
|
|
|
|
def _make_runner(stt_enabled: bool = True) -> "GatewayRunner": # type: ignore[name-defined]
|
|
from gateway.run import GatewayRunner
|
|
|
|
runner = GatewayRunner.__new__(GatewayRunner)
|
|
runner.config = GatewayConfig(stt_enabled=stt_enabled)
|
|
runner.adapters = {}
|
|
runner._model = "test-model"
|
|
runner._base_url = ""
|
|
runner._has_setup_skill = lambda: False
|
|
return runner
|
|
|
|
|
|
def _voice_event(path: str = "/tmp/voice.ogg") -> MessageEvent:
|
|
return MessageEvent(
|
|
text="",
|
|
message_type=MessageType.VOICE,
|
|
source=SessionSource(platform=Platform.TELEGRAM, chat_id="1", chat_type="dm"),
|
|
media_urls=[path],
|
|
media_types=["audio/ogg"],
|
|
)
|
|
|
|
|
|
def _audio_event(path: str = "/tmp/song.mp3") -> MessageEvent:
|
|
return MessageEvent(
|
|
text="",
|
|
message_type=MessageType.AUDIO,
|
|
source=SessionSource(platform=Platform.TELEGRAM, chat_id="1", chat_type="dm"),
|
|
media_urls=[path],
|
|
media_types=["audio/mpeg"],
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# 1. VOICE still goes through STT
|
|
# ---------------------------------------------------------------------------
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_voice_message_still_transcribed():
|
|
"""MessageType.VOICE must still be sent through _enrich_message_with_transcription."""
|
|
runner = _make_runner(stt_enabled=True)
|
|
source = SessionSource(platform=Platform.TELEGRAM, chat_id="1", chat_type="dm")
|
|
event = _voice_event("/tmp/voice.ogg")
|
|
|
|
with patch(
|
|
"tools.transcription_tools.transcribe_audio",
|
|
return_value={"success": True, "transcript": "hello world", "provider": "whisper"},
|
|
) as mock_transcribe:
|
|
result = await runner._prepare_inbound_message_text(
|
|
event=event,
|
|
source=source,
|
|
history=[],
|
|
)
|
|
|
|
mock_transcribe.assert_called_once_with("/tmp/voice.ogg", None, "gateway")
|
|
# The transcript passes through as a plain quoted line — no "voice message"
|
|
# meta-commentary in the LLM-visible prompt.
|
|
assert "hello world" in result
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# 2. AUDIO file attachment bypasses STT
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_audio_attachment_context_note_format():
|
|
"""Context note for audio file attachments should include the file path and guidance."""
|
|
runner = _make_runner(stt_enabled=True)
|
|
source = SessionSource(platform=Platform.TELEGRAM, chat_id="1", chat_type="dm")
|
|
event = _audio_event("/tmp/cache_12345_my_song.mp3")
|
|
|
|
with patch(
|
|
"tools.transcription_tools.transcribe_audio",
|
|
side_effect=AssertionError("must not be called"),
|
|
):
|
|
with patch(
|
|
"tools.credential_files.to_agent_visible_cache_path",
|
|
side_effect=lambda p: p,
|
|
):
|
|
result = await runner._prepare_inbound_message_text(
|
|
event=event,
|
|
source=source,
|
|
history=[],
|
|
)
|
|
|
|
assert "my_song.mp3" in result
|
|
assert "audio file attachment" in result.lower()
|
|
# Should NOT contain the voice-message transcription wrapper text
|
|
assert "voice message" not in result.lower()
|
|
# Guides the agent to transcribe/process the file itself rather than
|
|
# punting back to the user (same bug class as the PDF/DOCX note).
|
|
assert "transcri" in result.lower()
|
|
assert "ask the user what they'd like" not in result.lower()
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# 3. STT disabled still results in no transcription for audio file attachments
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# 4. Telegram gateway: msg.audio → MessageType.AUDIO (not VOICE)
|
|
# ---------------------------------------------------------------------------
|
|
|