From 36838b44d4133c426bc88a6776fee4d77d028744 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F=28Relakkes?= =?UTF-8?q?=29?= Date: Mon, 5 Oct 2026 21:00:03 +0800 Subject: [PATCH] feat(desktop): turn dictation into a recording bar and offer it by default Recording: - While dictating, the composer's toolbar row becomes a recording bar: cancel, a scrolling trace of what the microphone heard, the clock (a countdown in the last 15 s), stop, and send. Recording is no longer painted in the error color. - Stop writes the text into the draft. Send writes it, then submits the draft through the composer's own path once the render carries the text. Text held back because the draft changed is never sent. - Dictated text is tinted for a moment where it landed. - The settings transcription test uses the same trace, clock and stop disc. VoiceWave is removed. Default: - Voice input is on by default for preferences that never saved it; a saved `false` is kept. - In the desktop app the microphone shows before the model is downloaded and opens Settings > Voice input. The browser (H5) never offers it. --- .../src/components/chat/ChatInput.test.tsx | 54 ++- desktop/src/components/chat/ChatInput.tsx | 10 + .../src/components/chat/MentionComposer.tsx | 51 ++- desktop/src/components/ui/IconButton.test.tsx | 10 + desktop/src/components/ui/IconButton.tsx | 19 +- desktop/src/dev/ComponentGallery.tsx | 1 + .../voiceInput/VoiceInputButton.test.tsx | 371 +++++++++++++++++- .../features/voiceInput/VoiceInputButton.tsx | 140 +++---- .../features/voiceInput/VoiceRecordingBar.tsx | 206 ++++++++++ .../features/voiceInput/VoiceTrail.test.tsx | 235 +++++++++++ .../src/features/voiceInput/VoiceTrail.tsx | 217 ++++++++++ .../features/voiceInput/VoiceWave.test.tsx | 302 -------------- desktop/src/features/voiceInput/VoiceWave.tsx | 153 -------- .../voiceInput/useComposerDictation.ts | 84 +++- desktop/src/i18n/locales/en.ts | 7 + desktop/src/i18n/locales/jp.ts | 7 + desktop/src/i18n/locales/kr.ts | 7 + desktop/src/i18n/locales/zh-TW.ts | 7 + desktop/src/i18n/locales/zh.ts | 7 + desktop/src/pages/EmptySession.test.tsx | 36 ++ desktop/src/pages/EmptySession.tsx | 14 +- .../settings/VoiceInputSettings.test.tsx | 35 +- .../pages/settings/VoiceTranscriptionTest.tsx | 103 ++--- desktop/src/stores/voiceInputStore.test.ts | 18 + desktop/src/stores/voiceInputStore.ts | 11 + desktop/src/theme/globals.css | 28 ++ .../__tests__/desktop-ui-preferences.test.ts | 24 +- src/server/__tests__/voice-api.test.ts | 2 +- .../voice/__tests__/voiceService.test.ts | 4 +- src/server/services/voice/types.ts | 8 +- 30 files changed, 1512 insertions(+), 659 deletions(-) create mode 100644 desktop/src/features/voiceInput/VoiceRecordingBar.tsx create mode 100644 desktop/src/features/voiceInput/VoiceTrail.test.tsx create mode 100644 desktop/src/features/voiceInput/VoiceTrail.tsx delete mode 100644 desktop/src/features/voiceInput/VoiceWave.test.tsx delete mode 100644 desktop/src/features/voiceInput/VoiceWave.tsx diff --git a/desktop/src/components/chat/ChatInput.test.tsx b/desktop/src/components/chat/ChatInput.test.tsx index 5ddb179d..e9d5774b 100644 --- a/desktop/src/components/chat/ChatInput.test.tsx +++ b/desktop/src/components/chat/ChatInput.test.tsx @@ -3362,6 +3362,10 @@ describe('ChatInput file mentions', () => { beforeEach(() => { activeRecording = recording() armVoice() + // Voice input exists only in the desktop app. + window.desktopHost = { ...browserHost, kind: 'electron', isDesktop: true } + // jsdom has no canvas; the recording bar's trace draws nothing without one. + vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(null) }) it('puts the microphone between the model picker and the send button', () => { @@ -3384,6 +3388,12 @@ describe('ChatInput file mentions', () => { expect(screen.queryByTestId('voice-input')).toBeNull() }) + it('does not render the microphone in the browser (H5)', () => { + Reflect.deleteProperty(window, 'desktopHost') + render() + expect(screen.queryByTestId('voice-input')).toBeNull() + }) + it('writes dictated text at the caret without sending anything', async () => { render() setComposerText('ab', 1) @@ -3399,12 +3409,54 @@ describe('ChatInput file mentions', () => { expect(mocks.wsSend).not.toHaveBeenCalled() }) + it('hands the toolbar row to the recording bar and gives it back', async () => { + render() + await act(async () => { + fireEvent.click(screen.getByRole('button', { name: 'Dictate' })) + }) + + expect(screen.getByTestId('voice-recording-bar')).toBeVisible() + expect(screen.getByTestId('chat-input-toolbar-leading')).not.toBeVisible() + expect(screen.getByTestId('chat-input-toolbar-trailing')).not.toBeVisible() + // Hidden, not unmounted: the model picker keeps its own state. + expect(screen.getByTestId('model-selector-shell')).toBeInTheDocument() + + fireEvent.click(screen.getByRole('button', { name: 'Cancel recording (Esc)' })) + + expect(screen.queryByTestId('voice-recording-bar')).toBeNull() + expect(screen.getByTestId('chat-input-toolbar-trailing')).toBeVisible() + expect(screen.getByRole('button', { name: 'Dictate' })).toBeVisible() + }) + + it('sends the dictated text through the composer\'s own send path', async () => { + render() + await act(async () => { + fireEvent.click(screen.getByRole('button', { name: 'Dictate' })) + }) + await act(async () => { + fireEvent.click(await screen.findByRole('button', { name: 'Transcribe and send' })) + }) + expect(mocks.wsSend).not.toHaveBeenCalled() + + await act(async () => { + finishTranscription('你好') + }) + + expect(mocks.wsSend).toHaveBeenCalledWith(sessionId, expect.objectContaining({ + type: 'user_message', + content: '你好', + })) + expect(getComposerText()).toBe('') + }) + it('keeps the text aside when the message was sent while it was being recognised', async () => { + useSettingsStore.setState({ chatSendBehavior: 'enter' }) render() setComposerText('question') await dictate() - fireEvent.click(screen.getByRole('button', { name: 'Run' })) + // The send button is hidden behind the recording bar; Enter still sends. + fireEvent.keyDown(getComposerElement(), { key: 'Enter' }) expect(getComposerText()).toBe('') await act(async () => { finishTranscription('late words') diff --git a/desktop/src/components/chat/ChatInput.tsx b/desktop/src/components/chat/ChatInput.tsx index 07f94b85..60591830 100644 --- a/desktop/src/components/chat/ChatInput.tsx +++ b/desktop/src/components/chat/ChatInput.tsx @@ -80,6 +80,7 @@ import { getSessionWorkspaceState, getSessionSeedWorkDir } from '../../lib/sessi import { hasRunningSubagentTasks } from '../../lib/backgroundTasks' import { useComposerDictation } from '@/features/voiceInput/useComposerDictation' import { VoiceInputButton } from '@/features/voiceInput/VoiceInputButton' +import { VoiceRecordingBar } from '@/features/voiceInput/VoiceRecordingBar' type GitInfo = SessionGitInfo @@ -331,7 +332,13 @@ export function ChatInput({ variant = 'default', compact = false, sessionId, vis draft: input, blocked: composerDisabled, contextKey: visible ? activeTabId : null, + // Called from an effect, after this render has declared `handleSubmit`; + // it is the same path Enter takes, queueing behind a running turn. + onSubmit: () => { void handleSubmit() }, }) + // While dictating, the toolbar's controls stay mounted but hidden, and the + // recording bar takes their row. + const dictationLive = dictation.phase !== 'idle' const hasWorkspaceReferences = !isMemberSession && workspaceReferences.length > 0 const isHeroComposer = variant === 'hero' && !isMemberSession && !compact const resolvedWorkDir = activeSession?.workDir || gitInfo?.workDir || undefined @@ -1581,8 +1588,10 @@ export function ChatInput({ variant = 'default', compact = false, sessionId, vis data-testid="chat-input-toolbar" className={`flex min-w-0 items-center justify-between pt-1.5 ${isMobileComposer ? 'gap-1' : 'gap-2'}`} > + {dictationLive && }