Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions gateway/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -21966,6 +21966,20 @@ def _should_send_voice_reply(
)
return False

# Desktop speaks for itself: the desktop app's useAutoSpeakReplies
# hook calls playSpeechText for the same reply, so a gateway voice
# message here makes every auto-TTS reply play twice (two independent
# synthesizations of the same text — logged as one "Generating speech"
# plus two "TTS audio saved" entries). The desktop path is the
# authoritative one for that surface; the gateway has no way to know
# whether the client spoke, so it must stay out of the way (#90297).
if event.source.platform.value == "desktop":
logger.debug(
"Auto voice reply skipped: desktop surface speaks locally (mode=%s chat=%s)",
voice_mode, chat_id,
)
return False

# Dedup: agent already called TTS tool in THIS turn only
last_user_idx = None
for i, msg in enumerate(reversed(agent_messages)):
Expand Down
101 changes: 101 additions & 0 deletions tests/gateway/test_desktop_no_gateway_voice_reply.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
"""The gateway must not send an auto-TTS voice reply to the desktop surface (#90297).

When `voice.auto_tts` is on, two independent TTS paths fire for the same
reply on the desktop: the gateway's `_send_voice_reply` (via
`adapter.send_voice`) and the desktop app's `useAutoSpeakReplies` hook (via
`/api/audio/speak` → `playSpeechText`). Neither knows about the other, so
every reply plays twice (one "Generating speech", two "TTS audio saved").
The desktop hook is the authoritative speaker for that surface; the gateway
side now steps out.

The desktop platform value is produced by the session routing layer at
runtime (it is not a static ``Platform`` enum member), so the tests stub a
platform object whose ``.value`` is ``"desktop"`` — exactly the comparison
``_should_send_voice_reply`` performs.
"""

from unittest.mock import MagicMock

from gateway.config import Platform
from gateway.platforms.base import MessageEvent, MessageType
from gateway.run import GatewayRunner
from gateway.session import SessionSource


class _DesktopPlatform(str):
"""Hashable stand-in for the runtime desktop platform value.

The desktop platform is produced by the session routing layer at
runtime (not a static ``Platform`` enum member), so the tests stub a
hashable object whose ``.value`` is ``"desktop"`` — exactly the
comparison and dict-key usage ``_should_send_voice_reply`` performs.
"""

@property
def value(self) -> str:
return self


def _desktop_platform() -> _DesktopPlatform:
return _DesktopPlatform("desktop")


def _make_runner() -> GatewayRunner:
runner = GatewayRunner.__new__(GatewayRunner)
runner.adapters = {}
runner._voice_mode = {}
return runner


def _make_event(platform) -> MessageEvent:
return MessageEvent(
text="trigger",
source=SessionSource(
platform=platform,
chat_id="123",
user_id="u1",
user_name="User",
),
message_type=MessageType.TEXT,
message_id="456",
)


class TestDesktopNeverGetsGatewayVoiceReply:
def test_desktop_platform_skipped_even_with_auto_tts_on(self):
"""The #90297 shape: global auto_tts on and the desktop would
otherwise qualify — the gateway must still stay silent because the
desktop's useAutoSpeakReplies speaks the same reply itself."""
runner = _make_runner()
adapter = MagicMock()
adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
runner.adapters[_desktop_platform()] = adapter
runner._voice_mode = {} # no explicit mode: adapter_auto_tts governs

event = _make_event(_desktop_platform())
assert (
runner._should_send_voice_reply(event, "spoken twice?", [])
is False
)

def test_desktop_skipped_with_explicit_all_mode(self):
"""Even an explicit /voice all for the chat must not double-speak on
the desktop — the hook fires on every reply regardless of mode."""
runner = _make_runner()
runner.adapters["desktop"] = MagicMock()
runner._voice_mode = {("desktop", "123"): "all"}

event = _make_event(_desktop_platform())
assert runner._should_send_voice_reply(event, "hello", []) is False

def test_other_platforms_still_send(self):
"""Guard: the skip is desktop-only — Telegram with auto_tts on still
gets its gateway voice reply."""
runner = _make_runner()
adapter = MagicMock()
adapter._should_auto_tts_for_chat = MagicMock(return_value=True)
runner.adapters[Platform.TELEGRAM] = adapter
runner._voice_mode = {}

event = _make_event(Platform.TELEGRAM)
assert runner._should_send_voice_reply(event, "hello", []) is True
Loading