From c31c27e03a0f61eccb003c714aa8c59e809d44bd Mon Sep 17 00:00:00 2001 From: Carbon Date: Thu, 30 Jul 2026 04:06:11 +0300 Subject: [PATCH] fix(desktop): preserve voice stop across speech setup --- .../use-voice-conversation-rearm.test.tsx | 258 ++++++++++++++++++ .../composer/hooks/use-voice-conversation.ts | 41 ++- 2 files changed, 295 insertions(+), 4 deletions(-) create mode 100644 apps/desktop/src/app/chat/composer/hooks/use-voice-conversation-rearm.test.tsx diff --git a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation-rearm.test.tsx b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation-rearm.test.tsx new file mode 100644 index 00000000000..e993abc5730 --- /dev/null +++ b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation-rearm.test.tsx @@ -0,0 +1,258 @@ +import { act, cleanup, renderHook, waitFor } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { $voicePlayback } from '@/store/voice-playback' + +import { useVoiceConversation } from './use-voice-conversation' + +const mocks = vi.hoisted(() => { + let deferStreamStart = false + let onSilence: null | (() => void) = null + let resolveStreamStart: null | (() => void) = null + let resolveSpeech: null | ((outcome: 'done' | 'fallback') => void) = null + let streamAvailable = true + + const stopVoicePlayback = vi.fn(() => { + const current = $voicePlayback.get() + $voicePlayback.set({ ...current, sequence: current.sequence + 1, status: 'idle' }) + }) + + const playSpeechText = vi.fn(() => { + stopVoicePlayback() + + return Promise.resolve(true) + }) + + const handle = { + cancel: vi.fn(), + start: vi.fn(async (options: { onSilence: () => void }) => { + onSilence = options.onSilence + }), + stop: vi.fn(async () => ({ + audio: new Blob(['voice'], { type: 'audio/webm' }), + heardSpeech: true + })) + } + + return { + continueStreamStart() { + resolveStreamStart?.() + resolveStreamStart = null + }, + deferStreamStart() { + deferStreamStart = true + }, + finishSpeech(outcome: 'done' | 'fallback') { + resolveSpeech?.(outcome) + }, + handle, + playSpeechText, + resetSpeechMocks() { + deferStreamStart = false + resolveStreamStart = null + resolveSpeech = null + streamAvailable = true + }, + startSpeechStream: vi.fn(async () => { + if (deferStreamStart) { + await new Promise(resolve => { + resolveStreamStart = resolve + }) + } + + if (!streamAvailable) { + return null + } + + const current = $voicePlayback.get() + $voicePlayback.set({ ...current, sequence: current.sequence + 1, status: 'preparing' }) + + return { + append: vi.fn(), + done: new Promise<'done' | 'fallback'>(resolve => { + resolveSpeech = resolve + }), + finish: vi.fn() + } + }), + stopVoicePlayback, + triggerSilence() { + onSilence?.() + }, + useFallbackSpeech() { + streamAvailable = false + } + } +}) + +vi.mock('./use-mic-recorder', () => ({ + useMicRecorder: () => ({ handle: mocks.handle, level: 0 }) +})) + +vi.mock('@/lib/voice-barge-in', () => ({ + monitorSpeechDuringPlayback: () => vi.fn() +})) + +vi.mock('@/lib/voice-playback', () => ({ + markVoicePlaybackInterrupted: vi.fn(), + playSpeechText: mocks.playSpeechText, + startSpeechStream: mocks.startSpeechStream, + stopVoicePlayback: mocks.stopVoicePlayback +})) + +vi.mock('@/lib/thinking-sound', () => ({ + startThinkingSound: vi.fn(), + stopThinkingSound: vi.fn() +})) + +vi.mock('@/store/notifications', () => ({ + notify: vi.fn(), + notifyError: vi.fn() +})) + +vi.mock('@/i18n', () => ({ + useI18n: () => ({ + t: { + notifications: { + voice: { + configureSpeechToText: '', + couldNotStartSession: '', + microphoneFailed: '', + playbackFailed: '', + transcriptionFailed: '', + unavailable: '' + } + } + } + }) +})) + +function renderRearmConversation(responseId: string, responseText: string) { + let response: null | { id: string; pending: boolean; text: string } = null + + return renderHook( + ({ enabled }) => + useVoiceConversation({ + busy: false, + consumePendingResponse: vi.fn(), + enabled, + onSubmit: async () => { + response = { id: responseId, pending: false, text: responseText } + }, + onTranscribeAudio: async () => 'Hello', + pendingResponse: () => response + }), + { initialProps: { enabled: false } } + ) +} + +async function beginReply(hook: ReturnType) { + hook.rerender({ enabled: true }) + await waitFor(() => expect(mocks.handle.start).toHaveBeenCalledTimes(1)) + + await act(async () => { + mocks.triggerSilence() + }) +} + +describe('useVoiceConversation playback rearm', () => { + afterEach(() => { + cleanup() + vi.clearAllMocks() + mocks.resetSpeechMocks() + $voicePlayback.set({ + audioElement: null, + messageId: null, + sequence: 0, + source: null, + status: 'idle' + }) + }) + + it('re-arms the microphone after normal streaming playback completes', async () => { + $voicePlayback.set({ + audioElement: null, + messageId: null, + sequence: 7, + source: null, + status: 'idle' + }) + const hook = renderRearmConversation('reply-1', 'Hello back') + + await beginReply(hook) + await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled()) + expect($voicePlayback.get().sequence).toBeGreaterThan(7) + + await act(async () => { + mocks.finishSpeech('done') + }) + + await waitFor(() => expect(mocks.handle.start).toHaveBeenCalledTimes(2)) + expect(hook.result.current.status).toBe('listening') + }) + + it('honors Stop while streaming playback is still preparing', async () => { + mocks.deferStreamStart() + const hook = renderRearmConversation('reply-preparing', 'Do not play this') + + await beginReply(hook) + await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled()) + + mocks.stopVoicePlayback() + await act(async () => { + mocks.continueStreamStart() + }) + + await waitFor(() => expect(hook.result.current.status).toBe('idle')) + expect(mocks.stopVoicePlayback).toHaveBeenCalledTimes(2) + expect(mocks.handle.start).toHaveBeenCalledTimes(1) + }) + + it('does not start fallback playback after Stop during stream discovery', async () => { + mocks.deferStreamStart() + mocks.useFallbackSpeech() + const hook = renderRearmConversation('reply-no-stream', 'Do not fall back') + + await beginReply(hook) + await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled()) + + mocks.stopVoicePlayback() + await act(async () => { + mocks.continueStreamStart() + }) + + await waitFor(() => expect(hook.result.current.status).toBe('idle')) + expect(mocks.playSpeechText).not.toHaveBeenCalled() + expect(mocks.handle.start).toHaveBeenCalledTimes(1) + }) + + it('does not re-arm after an external Stop during streaming playback', async () => { + const hook = renderRearmConversation('reply-stopped', 'Playing now') + + await beginReply(hook) + await waitFor(() => expect(mocks.startSpeechStream).toHaveBeenCalled()) + + mocks.stopVoicePlayback() + await act(async () => { + mocks.finishSpeech('done') + }) + + await waitFor(() => expect(hook.result.current.status).toBe('idle')) + expect(mocks.handle.start).toHaveBeenCalledTimes(1) + }) + + it('re-arms the microphone after normal fallback playback completes', async () => { + mocks.useFallbackSpeech() + const hook = renderRearmConversation('reply-fallback', 'Fallback reply') + + await beginReply(hook) + + await waitFor(() => + expect(mocks.playSpeechText).toHaveBeenCalledWith('Fallback reply', { + source: 'voice-conversation' + }) + ) + await waitFor(() => expect(mocks.handle.start).toHaveBeenCalledTimes(2)) + expect(hook.result.current.status).toBe('listening') + }) +}) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts index 94fb912ff1c..5f4be4b637e 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-voice-conversation.ts @@ -262,7 +262,7 @@ export function useVoiceConversation({ }, [handle, handleTurn, onFatalError, voiceCopy.couldNotStartSession, voiceCopy.microphoneFailed]) const settleAfterSpeech = useCallback( - (barged: boolean) => { + (barged: boolean, stoppedDuringSetup = false) => { if (barged || !awaitingSpokenResponseRef.current) { awaitingSpokenResponseRef.current = false consumePendingResponse() @@ -286,7 +286,9 @@ export function useVoiceConversation({ // voice-playback sequence has advanced past what we captured at speech // start — don't auto-start the next sentence, the user chose to stop. const stoppedByUser = - speechStartSequenceRef.current > 0 && $voicePlayback.get().sequence > speechStartSequenceRef.current + stoppedDuringSetup || + (speechStartSequenceRef.current > 0 && + $voicePlayback.get().sequence > speechStartSequenceRef.current) speechStartSequenceRef.current = 0 @@ -458,9 +460,13 @@ export function useVoiceConversation({ // this is a safety net for read-aloud-style entries into the loop. ensureBargeMonitor() + const playback = playSpeechText(response.text, { source: 'voice-conversation' }) + // playSpeechText performs its normal cleanup synchronously before + // returning. Capture the sequence after that internal increment so + // only a later, external stop suppresses the next listen cycle. speechStartSequenceRef.current = $voicePlayback.get().sequence - void playSpeechText(response.text, { source: 'voice-conversation' }) + void playback .catch(error => notifyError(error, voiceCopy.playbackFailed)) .finally(() => { if (responseIdRef.current === responseId) { @@ -482,9 +488,10 @@ export function useVoiceConversation({ */ const openLiveSpeech = useCallback( (responseId: string) => { + const sequenceBeforeStart = $voicePlayback.get().sequence + responseIdRef.current = responseId spokenSourceLengthRef.current = 0 - speechStartSequenceRef.current = $voicePlayback.get().sequence setStatus('speaking') // VAD barge-in: the user talking over the reply cuts playback, drops @@ -506,6 +513,16 @@ export function useVoiceConversation({ } if (!session) { + // Stream discovery can also fail after an explicit Stop landed + // during its async URL lookup. In that case, do not turn the stopped + // live attempt into fresh fallback playback. + if ($voicePlayback.get().sequence > sequenceBeforeStart) { + awaitingSpokenResponseRef.current = false + settleAfterSpeech(false, true) + + return + } + // No streaming backend/provider: speak the whole reply once it lands. speechSessionRef.current = null awaitFallbackSpeech(responseId) @@ -513,8 +530,24 @@ export function useVoiceConversation({ return } + // startSpeechStream calls stopVoicePlayback once after its async URL + // lookup. A second sequence bump means the user pressed Stop while + // setup was still pending. Do not absorb that explicit stop into the + // post-start baseline or allow the new session to play. + const sequenceAfterStart = $voicePlayback.get().sequence + const stoppedDuringStart = sequenceAfterStart > sequenceBeforeStart + 1 + + speechStartSequenceRef.current = sequenceAfterStart speechSessionRef.current = session + if (stoppedDuringStart) { + stopVoicePlayback() + awaitingSpokenResponseRef.current = false + settleAfterSpeech(false, true) + + return + } + // Timer-driven feed: reply text flows into the session at delta rate // regardless of React render cadence. const feedTimer = window.setInterval(() => feedSpeechSession(responseId), 150)