diff --git a/desktop/src/components/chat/ChatInput.test.tsx b/desktop/src/components/chat/ChatInput.test.tsx
index 5ddb179d..e9d5774b 100644
--- a/desktop/src/components/chat/ChatInput.test.tsx
+++ b/desktop/src/components/chat/ChatInput.test.tsx
@@ -3362,6 +3362,10 @@ describe('ChatInput file mentions', () => {
beforeEach(() => {
activeRecording = recording()
armVoice()
+ // Voice input exists only in the desktop app.
+ window.desktopHost = { ...browserHost, kind: 'electron', isDesktop: true }
+ // jsdom has no canvas; the recording bar's trace draws nothing without one.
+ vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(null)
})
it('puts the microphone between the model picker and the send button', () => {
@@ -3384,6 +3388,12 @@ describe('ChatInput file mentions', () => {
expect(screen.queryByTestId('voice-input')).toBeNull()
})
+ it('does not render the microphone in the browser (H5)', () => {
+ Reflect.deleteProperty(window, 'desktopHost')
+ render()
+ expect(screen.queryByTestId('voice-input')).toBeNull()
+ })
+
it('writes dictated text at the caret without sending anything', async () => {
render()
setComposerText('ab', 1)
@@ -3399,12 +3409,54 @@ describe('ChatInput file mentions', () => {
expect(mocks.wsSend).not.toHaveBeenCalled()
})
+ it('hands the toolbar row to the recording bar and gives it back', async () => {
+ render()
+ await act(async () => {
+ fireEvent.click(screen.getByRole('button', { name: 'Dictate' }))
+ })
+
+ expect(screen.getByTestId('voice-recording-bar')).toBeVisible()
+ expect(screen.getByTestId('chat-input-toolbar-leading')).not.toBeVisible()
+ expect(screen.getByTestId('chat-input-toolbar-trailing')).not.toBeVisible()
+ // Hidden, not unmounted: the model picker keeps its own state.
+ expect(screen.getByTestId('model-selector-shell')).toBeInTheDocument()
+
+ fireEvent.click(screen.getByRole('button', { name: 'Cancel recording (Esc)' }))
+
+ expect(screen.queryByTestId('voice-recording-bar')).toBeNull()
+ expect(screen.getByTestId('chat-input-toolbar-trailing')).toBeVisible()
+ expect(screen.getByRole('button', { name: 'Dictate' })).toBeVisible()
+ })
+
+ it('sends the dictated text through the composer\'s own send path', async () => {
+ render()
+ await act(async () => {
+ fireEvent.click(screen.getByRole('button', { name: 'Dictate' }))
+ })
+ await act(async () => {
+ fireEvent.click(await screen.findByRole('button', { name: 'Transcribe and send' }))
+ })
+ expect(mocks.wsSend).not.toHaveBeenCalled()
+
+ await act(async () => {
+ finishTranscription('你好')
+ })
+
+ expect(mocks.wsSend).toHaveBeenCalledWith(sessionId, expect.objectContaining({
+ type: 'user_message',
+ content: '你好',
+ }))
+ expect(getComposerText()).toBe('')
+ })
+
it('keeps the text aside when the message was sent while it was being recognised', async () => {
+ useSettingsStore.setState({ chatSendBehavior: 'enter' })
render()
setComposerText('question')
await dictate()
- fireEvent.click(screen.getByRole('button', { name: 'Run' }))
+ // The send button is hidden behind the recording bar; Enter still sends.
+ fireEvent.keyDown(getComposerElement(), { key: 'Enter' })
expect(getComposerText()).toBe('')
await act(async () => {
finishTranscription('late words')
diff --git a/desktop/src/components/chat/ChatInput.tsx b/desktop/src/components/chat/ChatInput.tsx
index 07f94b85..60591830 100644
--- a/desktop/src/components/chat/ChatInput.tsx
+++ b/desktop/src/components/chat/ChatInput.tsx
@@ -80,6 +80,7 @@ import { getSessionWorkspaceState, getSessionSeedWorkDir } from '../../lib/sessi
import { hasRunningSubagentTasks } from '../../lib/backgroundTasks'
import { useComposerDictation } from '@/features/voiceInput/useComposerDictation'
import { VoiceInputButton } from '@/features/voiceInput/VoiceInputButton'
+import { VoiceRecordingBar } from '@/features/voiceInput/VoiceRecordingBar'
type GitInfo = SessionGitInfo
@@ -331,7 +332,13 @@ export function ChatInput({ variant = 'default', compact = false, sessionId, vis
draft: input,
blocked: composerDisabled,
contextKey: visible ? activeTabId : null,
+ // Called from an effect, after this render has declared `handleSubmit`;
+ // it is the same path Enter takes, queueing behind a running turn.
+ onSubmit: () => { void handleSubmit() },
})
+ // While dictating, the toolbar's controls stay mounted but hidden, and the
+ // recording bar takes their row.
+ const dictationLive = dictation.phase !== 'idle'
const hasWorkspaceReferences = !isMemberSession && workspaceReferences.length > 0
const isHeroComposer = variant === 'hero' && !isMemberSession && !compact
const resolvedWorkDir = activeSession?.workDir || gitInfo?.workDir || undefined
@@ -1581,8 +1588,10 @@ export function ChatInput({ variant = 'default', compact = false, sessionId, vis
data-testid="chat-input-toolbar"
className={`flex min-w-0 items-center justify-between pt-1.5 ${isMobileComposer ? 'gap-1' : 'gap-2'}`}
>
+ {dictationLive && }
{!isMemberSession && (
@@ -1672,6 +1681,7 @@ export function ChatInput({ variant = 'default', compact = false, sessionId, vis
{!isMemberSession && activeTabId && (
diff --git a/desktop/src/components/chat/MentionComposer.tsx b/desktop/src/components/chat/MentionComposer.tsx
index 7415d6b1..89d97526 100644
--- a/desktop/src/components/chat/MentionComposer.tsx
+++ b/desktop/src/components/chat/MentionComposer.tsx
@@ -48,8 +48,11 @@ export type MentionComposerHandle = {
* Replaces the projected-text range with plain text as one editor
* transaction, so a single undo removes it (unlike a `value` rewrite, which
* resets history). The caret lands after the inserted text.
+ *
+ * `flash` tints the new text for a moment, for writes the user did not type
+ * (dictation) and so has to find in the draft.
*/
- insertTextAtOffsets: (start: number, end: number, text: string) => void
+ insertTextAtOffsets: (start: number, end: number, text: string, options?: { flash?: boolean }) => void
/**
* Content for the model: text with each mention pill serialized to
* file paths or explicit skill/plugin requests. Read from the live document, so literal text that
@@ -85,6 +88,36 @@ export type MentionComposerProps = {
*/
const composerViewRegistry = new WeakMap()
const workflowKeywordPluginKey = new PluginKey('workflow-keyword-highlight')
+const insertionFlashPluginKey = new PluginKey('insertion-flash')
+/** Matches `composer-insertion-flash` in globals.css. */
+export const INSERTION_FLASH_MS = 1600
+
+/**
+ * Holds the one range `insertTextAtOffsets({ flash })` marked. Any other edit
+ * drops it: a tint that follows the text around after the user starts typing
+ * reads as a selection, not as "this is what just arrived".
+ */
+function insertionFlashPlugin() {
+ return new Plugin({
+ key: insertionFlashPluginKey,
+ state: {
+ init: () => DecorationSet.empty,
+ apply: (tr, current) => {
+ const range = tr.getMeta(insertionFlashPluginKey) as { from: number; to: number } | null | undefined
+ if (range === null) return DecorationSet.empty
+ if (range) {
+ return DecorationSet.create(tr.doc, [
+ Decoration.inline(range.from, range.to, { class: 'composer-insertion-flash' }),
+ ])
+ }
+ return tr.docChanged ? DecorationSet.empty : current
+ },
+ },
+ props: {
+ decorations: (state) => insertionFlashPluginKey.getState(state),
+ },
+ })
+}
export function getComposerViewForTesting(element: HTMLElement | null): EditorView | undefined {
return element ? composerViewRegistry.get(element) : undefined
@@ -138,6 +171,7 @@ export const MentionComposer = forwardRef(null)
const viewRef = useRef(null)
+ const flashTimerRef = useRef | null>(null)
const workflowKeywordTriggerEnabled = useSettingsStore(
(state) => state.workflowKeywordTriggerEnabled,
)
@@ -214,6 +248,7 @@ export const MentionComposer = forwardRef {
+ if (flashTimerRef.current) clearTimeout(flashTimerRef.current)
composerViewRegistry.delete(view.dom)
viewRef.current = null
view.destroy()
@@ -351,7 +387,7 @@ export const MentionComposer = forwardRef viewRef.current?.hasFocus() ?? false,
- insertTextAtOffsets: (start, end, text) => {
+ insertTextAtOffsets: (start, end, text, options) => {
const view = viewRef.current
if (!view || !text) return
const docLength = projectedDocLength(view.state.doc)
@@ -359,7 +395,18 @@ export const MentionComposer = forwardRef {
+ flashTimerRef.current = null
+ const current = viewRef.current
+ if (!current || insertionFlashPluginKey.getState(current.state) === DecorationSet.empty) return
+ current.dispatch(current.state.tr
+ .setMeta(insertionFlashPluginKey, null)
+ .setMeta('addToHistory', false))
+ }, INSERTION_FLASH_MS)
},
getModelContent: () => {
const view = viewRef.current
diff --git a/desktop/src/components/ui/IconButton.test.tsx b/desktop/src/components/ui/IconButton.test.tsx
index e5f5bf04..97c1e3e0 100644
--- a/desktop/src/components/ui/IconButton.test.tsx
+++ b/desktop/src/components/ui/IconButton.test.tsx
@@ -205,6 +205,16 @@ describe('IconButton', () => {
expect(container.firstElementChild?.className).not.toContain('hover:bg-[var(--color-error-soft)]')
})
+ it('soft rests on a neutral disc with no border, and owns the only hover fill', () => {
+ // `filled` is a bordered card face; the recording bar's cancel and stop sit
+ // beside a waveform and need a ground without becoming separate objects.
+ const { container } = render(} label="Cancel" tone="default" soft filled />)
+ const classes = container.firstElementChild!.className.split(/\s+/)
+ expect(classes).toContain('bg-[var(--color-btn-soft-bg)]')
+ expect(classes.filter((c) => c.startsWith('hover:bg-'))).toEqual(['hover:bg-[var(--color-btn-soft-hover)]'])
+ expect(classes.some((c) => c === 'border' || c.startsWith('border-'))).toBe(false)
+ })
+
it('emits exactly one disabled opacity', () => {
// Same trap as the hover text: a caller passing `disabled:opacity-0` via
// className loses to the component's own `disabled:opacity-50`, because
diff --git a/desktop/src/components/ui/IconButton.tsx b/desktop/src/components/ui/IconButton.tsx
index b602ac87..2de1f792 100644
--- a/desktop/src/components/ui/IconButton.tsx
+++ b/desktop/src/components/ui/IconButton.tsx
@@ -44,6 +44,13 @@ export type IconButtonProps =
* washes out against a photo.
*/
solid?: boolean
+ /**
+ * A neutral resting disc with no border, for controls that share a strip
+ * with something busy and need their own ground: the composer's recording
+ * bar sets cancel and stop beside a live waveform. `filled` is a bordered
+ * card face, which reads as a separate object instead.
+ */
+ soft?: boolean
/** Hairline border on a transparent background. */
bordered?: boolean
/**
@@ -212,6 +219,9 @@ const SOLID_CLASSES: Record = {
danger: 'bg-[var(--color-error)] text-[var(--color-on-error)]',
}
+/** `soft`: its own resting fill and the hover that steps past it. */
+const SOFT_CLASSES = 'bg-[var(--color-btn-soft-bg)] hover:bg-[var(--color-btn-soft-hover)]'
+
const PRESSED_CLASSES: Record = {
default: 'bg-[var(--color-surface-selected)] text-[var(--color-text-primary)]',
sidebar: 'bg-[var(--color-sidebar-item-hover)] text-[var(--color-text-primary)]',
@@ -251,6 +261,7 @@ export const IconButton = forwardRef(functio
shape = 'square',
filled = false,
solid = false,
+ soft = false,
bordered = false,
hoverTone,
pressed,
@@ -281,9 +292,9 @@ export const IconButton = forwardRef(functio
// color and hover fill are skipped rather than left to compete.
solid ? SOLID_CLASSES[tone] : (SURFACE_REST_TEXT[surface] ?? REST_TEXT)[tone],
solid && 'hover:brightness-110',
- // A pressed button carries its own fill and hover; skipping the tone's
- // hover here keeps two `hover:bg-[…]` values from competing.
- !solid && (pressed ? PRESSED_CLASSES[surface] : HOVER_BG[surface][tone]),
+ // A pressed or soft button carries its own fill and hover; skipping the
+ // tone's hover here keeps two `hover:bg-[…]` values from competing.
+ !solid && (pressed ? PRESSED_CLASSES[surface] : soft ? SOFT_CLASSES : HOVER_BG[surface][tone]),
// Exactly one hover text color — `hoverTone` replaces the tone's own
// rather than stacking on top of it, which Tailwind would silently
// resolve the wrong way.
@@ -292,7 +303,7 @@ export const IconButton = forwardRef(functio
? 'hover:text-[var(--color-error)]'
: (SURFACE_HOVER_TEXT[surface] ?? HOVER_TEXT)[tone]
),
- filled && !pressed && !solid && FILLED_CLASSES[tone],
+ filled && !pressed && !solid && !soft && FILLED_CLASSES[tone],
bordered && !filled && 'border border-[var(--color-border)]',
shape === 'circle' && 'rounded-full',
className,
diff --git a/desktop/src/dev/ComponentGallery.tsx b/desktop/src/dev/ComponentGallery.tsx
index a8a234b2..bfcdfd8a 100644
--- a/desktop/src/dev/ComponentGallery.tsx
+++ b/desktop/src/dev/ComponentGallery.tsx
@@ -563,6 +563,7 @@ export function ComponentGallery() {
))}
} label="Filled" tone={tone} filled />
} label="Bordered" tone={tone} bordered />
+ } label="Soft" tone={tone} shape="circle" soft />
} label="Circle" tone={tone} shape="circle" />
} label="Loading" tone={tone} loading />
diff --git a/desktop/src/features/voiceInput/VoiceInputButton.test.tsx b/desktop/src/features/voiceInput/VoiceInputButton.test.tsx
index 43aef437..bf12fc6e 100644
--- a/desktop/src/features/voiceInput/VoiceInputButton.test.tsx
+++ b/desktop/src/features/voiceInput/VoiceInputButton.test.tsx
@@ -5,14 +5,23 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import '@testing-library/jest-dom'
import { ApiError } from '@/api/client'
import type { VoiceCatalog } from '@/api/voice'
-import { getComposerViewForTesting, MentionComposer, type MentionComposerHandle } from '@/components/chat/MentionComposer'
+import {
+ getComposerViewForTesting,
+ INSERTION_FLASH_MS,
+ MentionComposer,
+ type MentionComposerHandle,
+} from '@/components/chat/MentionComposer'
import { translate } from '@/i18n'
+import { browserHost } from '@/lib/desktopHost/browserHost'
import type { TranslationKey } from '@/i18n/locales/en'
import { useSettingsStore } from '@/stores/settingsStore'
+import { SETTINGS_TAB_ID, useTabStore } from '@/stores/tabStore'
+import { useUIStore } from '@/stores/uiStore'
import { selectVoiceInputReady, useVoiceInputStore } from '@/stores/voiceInputStore'
import { VoiceRecorderError, type StartRecordingOptions } from './recorder'
import { useComposerDictation } from './useComposerDictation'
import { VoiceInputButton } from './VoiceInputButton'
+import { COUNTDOWN_SECONDS, VoiceRecordingBar } from './VoiceRecordingBar'
const mocks = vi.hoisted(() => ({
transcribe: vi.fn(),
@@ -41,7 +50,10 @@ vi.mock('./recorder', async (importOriginal) => ({
const en = (key: TranslationKey) => translate('en', key)
-function catalogFixture(overrides: Partial = {}, phase: 'ready' | 'unprepared' = 'ready'): VoiceCatalog {
+function catalogFixture(
+ overrides: Partial = {},
+ phase: 'ready' | 'unprepared' | 'downloading' = 'ready',
+): VoiceCatalog {
return {
supported: true,
providers: [{
@@ -74,16 +86,30 @@ let failTranscription: (error: unknown) => void
let transcribeSignal: AbortSignal | undefined
let handle: MentionComposerHandle | null
-type HarnessProps = { initial?: string; blocked?: boolean; contextKey?: string | null }
+type HarnessProps = {
+ initial?: string
+ blocked?: boolean
+ contextKey?: string | null
+ /** Receives the draft as the render that submits it sees it. */
+ onSubmit?: (draft: string) => void
+}
-function Harness({ initial = '', blocked = false, contextKey = 'session-a' }: HarnessProps) {
+/** Wired like ChatInput and EmptySession: the bar takes the toolbar's place while dictating. */
+function Harness({ initial = '', blocked = false, contextKey = 'session-a', onSubmit }: HarnessProps) {
const [input, setInput] = useState(initial)
const [mentions, setMentions] = useState([])
const ref = useRef(null)
- const dictation = useComposerDictation({ composerRef: ref, draft: input, blocked, contextKey })
+ const dictation = useComposerDictation({
+ composerRef: ref,
+ draft: input,
+ blocked,
+ contextKey,
+ onSubmit: onSubmit ? () => onSubmit(input) : undefined,
+ })
useEffect(() => {
handle = ref.current
})
+ const live = dictation.phase !== 'idle'
return (
-
+
+ {live && }
+
+
+
+
)
@@ -125,6 +156,9 @@ const draft = () => screen.getByTestId('draft').textContent
const startButton = () => screen.getByRole('button', { name: en('voice.composer.start') })
const stopButton = () => screen.getByRole('button', { name: en('voice.composer.stop') })
+const cancelButton = () => screen.getByRole('button', { name: en('voice.composer.cancel') })
+const sendButton = () => screen.getByRole('button', { name: en('voice.composer.sendNow') })
+const recordingBar = () => screen.queryByTestId('voice-recording-bar')
async function beginRecording() {
await act(async () => {
@@ -167,6 +201,10 @@ beforeEach(() => {
useSettingsStore.setState({ locale: 'en' })
useVoiceInputStore.setState({ catalog: catalogFixture(), loading: false, error: null })
localStorage.clear()
+ // Voice input exists only in the desktop app.
+ window.desktopHost = { ...browserHost, kind: 'electron', isDesktop: true }
+ // jsdom has no canvas; the trace draws nothing without a context.
+ vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(null)
// jsdom has no layout. ProseMirror reads Range geometry when a transaction
// scrolls the new selection into view.
Object.defineProperties(Range.prototype, {
@@ -178,6 +216,7 @@ beforeEach(() => {
afterEach(() => {
Reflect.deleteProperty(Range.prototype, 'getClientRects')
Reflect.deleteProperty(Range.prototype, 'getBoundingClientRect')
+ Reflect.deleteProperty(window, 'desktopHost')
cleanup()
vi.useRealTimers()
vi.restoreAllMocks()
@@ -191,7 +230,9 @@ describe('visibility', () => {
it.each([
['dictation is disabled', () => useVoiceInputStore.setState({ catalog: catalogFixture({ enabled: false }) })],
- ['the model is not downloaded', () => useVoiceInputStore.setState({ catalog: catalogFixture({}, 'unprepared') })],
+ // Plain-HTTP H5 has no microphone, HTTPS H5 is unverified on phones, and
+ // the H5 settings have no voice page: the browser never offers it.
+ ['running in the browser (H5), even with the model ready', () => Reflect.deleteProperty(window, 'desktopHost')],
['the platform does not support it', () => useVoiceInputStore.setState({ catalog: { ...catalogFixture(), supported: false } })],
['the environment cannot capture audio', () => mocks.supported.mockReturnValue(false)],
])('renders nothing and takes no space when %s', (_name, arrange) => {
@@ -211,7 +252,17 @@ describe('visibility', () => {
expect(await screen.findByRole('button', { name: en('voice.composer.start') })).toBeInTheDocument()
})
- it('keeps the button mounted mid-recording if the service is switched off meanwhile', async () => {
+ it('does not even ask for the catalog in the browser (H5)', () => {
+ Reflect.deleteProperty(window, 'desktopHost')
+ useVoiceInputStore.setState({ catalog: null })
+
+ render()
+
+ expect(mocks.catalog).not.toHaveBeenCalled()
+ expect(screen.queryByTestId('voice-input')).toBeNull()
+ })
+
+ it('keeps the recording controls mid-recording if the service is switched off meanwhile', async () => {
render()
await beginRecording()
@@ -220,10 +271,65 @@ describe('visibility', () => {
expect(stopButton()).toBeInTheDocument()
})
- it('does not steal focus from the composer on mouse down', () => {
+ it('does not steal focus from the composer on mouse down', async () => {
render()
// fireEvent returns false when the default action was prevented.
expect(fireEvent.mouseDown(startButton())).toBe(false)
+
+ await beginRecording()
+ for (const button of [cancelButton(), stopButton(), sendButton()]) {
+ expect(fireEvent.mouseDown(button)).toBe(false)
+ }
+ })
+})
+
+describe('before the model is downloaded, in the desktop app', () => {
+ const needsModelButton = () => screen.getByRole('button', { name: en('voice.composer.needsModel') })
+
+ beforeEach(() => {
+ useTabStore.setState({ tabs: [], activeTabId: null })
+ useUIStore.setState({ pendingSettingsTab: null })
+ useVoiceInputStore.setState({ catalog: catalogFixture({}, 'unprepared') })
+ })
+
+ it('shows the microphone, and a click opens Settings → Voice input instead of recording', () => {
+ render()
+
+ fireEvent.click(needsModelButton())
+
+ expect(mocks.startRecording).not.toHaveBeenCalled()
+ expect(useUIStore.getState().pendingSettingsTab).toBe('voice')
+ expect(useTabStore.getState().activeTabId).toBe(SETTINGS_TAB_ID)
+ expect(recordingBar()).toBeNull()
+ })
+
+ it('says the model is downloading while it is', () => {
+ useVoiceInputStore.setState({ catalog: catalogFixture({}, 'downloading') })
+ render()
+
+ fireEvent.click(screen.getByRole('button', { name: en('voice.composer.modelDownloading') }))
+
+ expect(useUIStore.getState().pendingSettingsTab).toBe('voice')
+ expect(mocks.startRecording).not.toHaveBeenCalled()
+ })
+
+ it('records with the same button as soon as the download finishes, without a reload', async () => {
+ render()
+ expect(needsModelButton()).toBeInTheDocument()
+
+ // The settings page and the composer share the store: its status poll lands here.
+ act(() => useVoiceInputStore.setState({ catalog: catalogFixture() }))
+
+ await beginRecording()
+ expect(mocks.startRecording).toHaveBeenCalledTimes(1)
+ expect(useTabStore.getState().activeTabId).toBeNull()
+ })
+
+ it('shows nothing once voice input is switched off', () => {
+ useVoiceInputStore.setState({ catalog: catalogFixture({ enabled: false }, 'unprepared') })
+ render()
+
+ expect(screen.queryByTestId('voice-input')).toBeNull()
})
})
@@ -310,31 +416,240 @@ describe('recording and transcription', () => {
})
describe('button semantics', () => {
- it('is a pressed toggle only while recording; starting and transcribing are busy, not pressed', async () => {
+ it('names the stop control for each phase; opening the microphone and recognition are busy', async () => {
let grant!: (value: FakeRecording) => void
mocks.startRecording.mockImplementation((options: StartRecordingOptions) => {
startOptions = options
return new Promise(resolve => { grant = resolve as typeof grant })
})
render()
- expect(startButton()).not.toHaveAttribute('aria-pressed')
await act(async () => {
fireEvent.click(startButton())
})
const starting = screen.getByRole('button', { name: en('voice.composer.starting') })
expect(starting).toHaveAttribute('aria-busy', 'true')
- expect(starting).not.toHaveAttribute('aria-pressed')
+ // Nothing has been recorded yet, so there is nothing to send.
+ expect(sendButton()).toBeDisabled()
await act(async () => {
grant(recording)
})
- expect(stopButton()).toHaveAttribute('aria-pressed', 'true')
+ expect(stopButton()).not.toHaveAttribute('aria-busy')
+ expect(sendButton()).toBeEnabled()
await stopRecording()
const transcribing = screen.getByRole('button', { name: en('voice.composer.transcribing') })
expect(transcribing).toHaveAttribute('aria-busy', 'true')
- expect(transcribing).not.toHaveAttribute('aria-pressed')
+ })
+})
+
+describe('recording bar', () => {
+ it("takes the toolbar's place while dictating and gives it back afterwards", async () => {
+ render()
+ expect(recordingBar()).toBeNull()
+
+ await beginRecording()
+ expect(recordingBar()).toBeVisible()
+ expect(screen.getByTestId('toolbar-controls')).not.toBeVisible()
+ expect(screen.queryByRole('button', { name: en('voice.composer.start') })).toBeNull()
+
+ await stopRecording()
+ expect(recordingBar()).toBeVisible()
+
+ await deliver('hello')
+ expect(recordingBar()).toBeNull()
+ expect(startButton()).toBeVisible()
+ })
+
+ it('cancels from its own button, without transcribing', async () => {
+ render()
+ await beginRecording()
+
+ fireEvent.click(cancelButton())
+
+ expect(recording.cancel).toHaveBeenCalledTimes(1)
+ expect(mocks.transcribe).not.toHaveBeenCalled()
+ expect(recordingBar()).toBeNull()
+ expect(draft()).toBe('keep')
+ })
+
+ it('cancels recognition from its own button and ignores a late answer', async () => {
+ render()
+ await beginRecording()
+ await stopRecording()
+
+ fireEvent.click(cancelButton())
+ expect(transcribeSignal?.aborted).toBe(true)
+ await deliver('too late')
+
+ expect(draft()).toBe('')
+ })
+
+ it('does not paint a recording in the error color: red is reserved for failures', async () => {
+ render()
+ await beginRecording()
+
+ expect(recordingBar()!.outerHTML).not.toMatch(/--color-error/)
+ })
+
+ it('shows recognition in place of the clock', async () => {
+ render()
+ await beginRecording()
+ expect(screen.getByTestId('voice-input-timer')).toHaveTextContent('0:00')
+
+ await stopRecording()
+ expect(screen.queryByTestId('voice-input-timer')).toBeNull()
+ expect(recordingBar()).toHaveTextContent(en('voice.composer.transcribing'))
+ })
+
+ it(`counts down the last ${COUNTDOWN_SECONDS} seconds before the limit`, async () => {
+ vi.useFakeTimers({ shouldAdvanceTime: true })
+ render()
+ await beginRecording()
+
+ // The fixture's limit is 60 s.
+ await act(async () => {
+ await vi.advanceTimersByTimeAsync(44_000)
+ })
+ const clock = screen.getByTestId('voice-input-timer')
+ expect(clock).toHaveTextContent(/^0:4[45]$/)
+ expect(clock).not.toHaveAttribute('data-countdown')
+
+ await act(async () => {
+ await vi.advanceTimersByTimeAsync(2_000)
+ })
+ expect(clock).toHaveAttribute('data-countdown', 'true')
+ expect(clock).toHaveTextContent(/^0:1[34] left$/)
+ })
+
+ it('hands focus to stop when started from the keyboard, and back to the draft when done', async () => {
+ render()
+ startButton().focus()
+ await beginRecording()
+ expect(stopButton()).toHaveFocus()
+
+ await stopRecording()
+ await deliver('hello')
+
+ expect(document.activeElement).toBe(editorView().editor)
+ })
+
+ it('leaves focus in the draft when dictation was started with the mouse', async () => {
+ render()
+ const { editor } = editorView()
+ editor.focus()
+ // A real click cannot move focus: the button prevents its mousedown.
+ fireEvent.mouseDown(startButton())
+ await beginRecording()
+
+ expect(editor).toHaveFocus()
+ })
+
+ it('does not pull a mouse user into the draft when nothing had focus', async () => {
+ // Focusing the editor would put its caret at the start, and a held result
+ // inserted later would land there instead of at the end.
+ render()
+ await beginRecording()
+ expect(document.activeElement).toBe(document.body)
+
+ await stopRecording()
+ await deliver('hello')
+
+ expect(document.activeElement).toBe(document.body)
+ })
+})
+
+describe('send from the recording bar', () => {
+ it('writes the text, then submits the draft that already carries it', async () => {
+ const onSubmit = vi.fn()
+ render()
+ await beginRecording()
+
+ await act(async () => {
+ fireEvent.click(sendButton())
+ })
+ expect(recording.stop).toHaveBeenCalledTimes(1)
+ expect(sendButton()).toHaveAttribute('aria-busy', 'true')
+ expect(recordingBar()).toHaveTextContent(en('voice.composer.transcribingThenSend'))
+ expect(onSubmit).not.toHaveBeenCalled()
+
+ await deliver('dictated')
+
+ expect(onSubmit).toHaveBeenCalledTimes(1)
+ expect(onSubmit).toHaveBeenCalledWith('hello dictated')
+ })
+
+ it('can still be chosen while the text is being recognised', async () => {
+ const onSubmit = vi.fn()
+ render()
+ await beginRecording()
+ await stopRecording()
+
+ await act(async () => {
+ fireEvent.click(sendButton())
+ })
+ expect(recording.stop).toHaveBeenCalledTimes(1)
+ await deliver('later')
+
+ expect(onSubmit).toHaveBeenCalledTimes(1)
+ expect(onSubmit).toHaveBeenCalledWith('later')
+ })
+
+ it('stop alone never sends', async () => {
+ const onSubmit = vi.fn()
+ render()
+ await beginRecording()
+ await stopRecording()
+ await deliver('just text')
+
+ expect(draft()).toBe('just text')
+ expect(onSubmit).not.toHaveBeenCalled()
+ })
+
+ it('never sends text it had to hold back because the draft changed', async () => {
+ const onSubmit = vi.fn()
+ render()
+ await beginRecording()
+ await act(async () => {
+ fireEvent.click(sendButton())
+ })
+ typeText(' typed', 5)
+ await deliver('dictated')
+
+ expect(screen.getByTestId('voice-input-pending-text')).toHaveTextContent('dictated')
+ expect(onSubmit).not.toHaveBeenCalled()
+
+ // Inserting the held text later is an edit, not a send.
+ fireEvent.click(screen.getByRole('button', { name: en('voice.composer.insertText') }))
+ expect(onSubmit).not.toHaveBeenCalled()
+ })
+
+ it('sends nothing when recognition finds no speech', async () => {
+ const onSubmit = vi.fn()
+ render()
+ await beginRecording()
+ await act(async () => {
+ fireEvent.click(sendButton())
+ })
+ await deliver(' ')
+
+ expect(screen.getByRole('alert')).toHaveTextContent(en('voice.composer.error.noSpeech'))
+ expect(onSubmit).not.toHaveBeenCalled()
+ })
+
+ it('a cancelled send does not fire on the next edit', async () => {
+ const onSubmit = vi.fn()
+ render()
+ await beginRecording()
+ await act(async () => {
+ fireEvent.click(sendButton())
+ })
+ fireEvent.click(cancelButton())
+ await deliver('too late')
+ typeText('typed', 0)
+
+ expect(onSubmit).not.toHaveBeenCalled()
})
})
@@ -496,6 +811,34 @@ describe('write-back position', () => {
expect(handle!.getSelectionOffsets()).toEqual({ start: 2, end: 2 })
})
+ it('marks the dictated text for a moment so the user can find it', async () => {
+ vi.useFakeTimers({ shouldAdvanceTime: true })
+ render()
+ await beginRecording()
+ await stopRecording()
+ await deliver('added')
+
+ const { editor } = editorView()
+ expect(editor.querySelector('.composer-insertion-flash')).toHaveTextContent('added')
+
+ await act(async () => {
+ await vi.advanceTimersByTimeAsync(INSERTION_FLASH_MS)
+ })
+ expect(editor.querySelector('.composer-insertion-flash')).toBeNull()
+ expect(draft()).toBe('keep added')
+ })
+
+ it('drops the mark as soon as the user edits', async () => {
+ render()
+ await beginRecording()
+ await stopRecording()
+ await deliver('added')
+
+ typeText('!', 0)
+
+ expect(editorView().editor.querySelector('.composer-insertion-flash')).toBeNull()
+ })
+
it('is one undoable editor step', async () => {
render()
await beginRecording()
diff --git a/desktop/src/features/voiceInput/VoiceInputButton.tsx b/desktop/src/features/voiceInput/VoiceInputButton.tsx
index fe88643b..94501cc3 100644
--- a/desktop/src/features/voiceInput/VoiceInputButton.tsx
+++ b/desktop/src/features/voiceInput/VoiceInputButton.tsx
@@ -1,10 +1,18 @@
-import { useEffect, useRef, useState } from 'react'
-import { Mic, Square, X } from 'lucide-react'
+import { useEffect, useState } from 'react'
+import { Mic, X } from 'lucide-react'
import { Button } from '@/components/ui/Button'
import { IconButton } from '@/components/ui/IconButton'
import { useTranslation } from '@/i18n'
import type { TranslationKey } from '@/i18n/locales/en'
-import { selectVoiceInputReady, useVoiceInputStore } from '@/stores/voiceInputStore'
+import { isDesktopRuntime } from '@/lib/desktopRuntime'
+import { SETTINGS_TAB_ID, useTabStore } from '@/stores/tabStore'
+import { useUIStore } from '@/stores/uiStore'
+import {
+ selectActiveVoiceProvider,
+ selectVoiceInputNeedsModel,
+ selectVoiceInputReady,
+ useVoiceInputStore,
+} from '@/stores/voiceInputStore'
import { isVoiceCaptureSupported } from './recorder'
import type { ComposerDictation, DictationIssue } from './useComposerDictation'
@@ -33,106 +41,60 @@ const ISSUE_KEYS: Record = {
/** Nothing was wrong; there was just nothing to write. */
const SOFT_ISSUES = new Set(['noSpeech', 'tooShort'])
-function formatElapsed(ms: number): string {
- const total = Math.max(0, Math.floor(ms / 1000))
- return `${Math.floor(total / 60)}:${String(total % 60).padStart(2, '0')}`
-}
-
/**
- * The composer's dictation control. Renders nothing until the voice service is
- * enabled, its model is downloaded, and this environment can capture audio.
+ * The composer's dictation control. Shown in the desktop app while voice input
+ * is switched on.
+ *
+ * Once the model is downloaded it starts a dictation and reports how the last
+ * one ended (an issue, or text held back from the draft); while one is under
+ * way the composer hides its toolbar, this button with it, and shows
+ * `VoiceRecordingBar` instead. Until then it opens Settings → Voice input,
+ * where the download is — the model is never fetched behind the user's back.
+ *
+ * Never in the browser (H5): over the usual plain-HTTP LAN address the page is
+ * not a secure context, so the browser offers no microphone at all; the HTTPS
+ * path has not been verified on phones; and the H5 settings have no voice page
+ * to send anyone to.
*/
export function VoiceInputButton({ dictation, blocked = false, mobile = false }: VoiceInputButtonProps) {
const t = useTranslation()
const ready = useVoiceInputStore(selectVoiceInputReady)
+ const needsModel = useVoiceInputStore(selectVoiceInputNeedsModel)
+ const modelPhase = useVoiceInputStore(state => selectActiveVoiceProvider(state)?.preparation.phase)
const loadCatalog = useVoiceInputStore(state => state.loadCatalog)
- const [supported] = useState(isVoiceCaptureSupported)
- const { phase, issue, pendingText, startedAt, getLevel } = dictation
- const haloRef = useRef(null)
- const [elapsed, setElapsed] = useState(0)
+ const [supported] = useState(() => isDesktopRuntime() && isVoiceCaptureSupported())
+ const { phase, issue, pendingText } = dictation
useEffect(() => {
- void loadCatalog()
- }, [loadCatalog])
-
- // Loudness drives the halo straight through the DOM: a 60 Hz value has no
- // business re-rendering the composer.
- useEffect(() => {
- if (phase !== 'recording') return
- let frame = 0
- const tick = () => {
- const halo = haloRef.current
- if (halo) {
- const level = getLevel()
- halo.style.transform = `scale(${1 + level * 0.6})`
- halo.style.opacity = String(0.25 + level * 0.75)
- }
- frame = requestAnimationFrame(tick)
- }
- frame = requestAnimationFrame(tick)
- return () => cancelAnimationFrame(frame)
- }, [getLevel, phase])
-
- useEffect(() => {
- if (phase !== 'recording') return
- setElapsed(0)
- const timer = setInterval(() => setElapsed(Date.now() - startedAt), 250)
- return () => clearInterval(timer)
- }, [phase, startedAt])
+ if (supported) void loadCatalog()
+ }, [loadCatalog, supported])
const engaged = phase !== 'idle' || pendingText !== null || issue !== null
- if (!supported || (!ready && !engaged)) return null
+ if (!supported || (!ready && !needsModel && !engaged)) return null
- const size = mobile ? '2xl' : 'md'
- const recording = phase === 'recording'
- const label = phase === 'idle'
+ const openVoiceSettings = () => {
+ useUIStore.getState().setPendingSettingsTab('voice')
+ useTabStore.getState().openTab(SETTINGS_TAB_ID, t('sidebar.settings'), 'settings')
+ }
+ const label = !needsModel
? t('voice.composer.start')
- : phase === 'starting'
- ? t('voice.composer.starting')
- : recording
- ? t('voice.composer.stop')
- : t('voice.composer.transcribing')
+ : modelPhase === 'downloading' || modelPhase === 'verifying'
+ ? t('voice.composer.modelDownloading')
+ : t('voice.composer.needsModel')
return (
-
- {recording && (
-
- {formatElapsed(elapsed)}
-
- )}
-
- {recording && (
-
- )}
-
- : }
- label={label}
- size={size}
- tone={recording ? 'danger' : 'secondary'}
- solid={recording}
- loading={phase === 'transcribing'}
- // A toggle only while it is recording. The label changes with the
- // phase, so pressed would contradict it during the other phases;
- // those are busy instead (`loading` already sets it for transcribing).
- pressed={recording ? true : undefined}
- aria-busy={phase === 'starting' || phase === 'transcribing' ? true : undefined}
- // Keep the caret in the composer: a click would otherwise blur it, and
- // the write-back position is the caret the user left there.
- onMouseDown={event => event.preventDefault()}
- onClick={dictation.toggle}
- className="relative"
- />
-
+
+ }
+ label={label}
+ size={mobile ? '2xl' : 'md'}
+ tone="secondary"
+ data-needs-model={needsModel || undefined}
+ // Keep the caret in the composer: a click would otherwise blur it, and
+ // the write-back position is the caret the user left there.
+ onMouseDown={event => event.preventDefault()}
+ onClick={needsModel ? openVoiceSettings : dictation.toggle}
+ />
{issue && (
Date.now())
+
+ useEffect(() => {
+ if (phase !== 'recording') return
+ setNow(Date.now())
+ const timer = setInterval(() => setNow(Date.now()), 250)
+ return () => clearInterval(timer)
+ }, [phase])
+
+ const className = 'shrink-0 whitespace-nowrap text-[13px] tabular-nums'
+ if (phase === 'transcribing') {
+ return {transcribingLabel}
+ }
+
+ const elapsed = phase === 'recording' ? Math.max(0, (now - startedAt) / 1000) : 0
+ const remaining = limitSeconds - elapsed
+ const countdown = phase === 'recording' && limitSeconds > 0 && remaining <= countdownSeconds(limitSeconds)
+ return (
+
+ {countdown
+ ? t('voice.composer.remaining', { time: formatClock(Math.ceil(Math.max(0, remaining))) })
+ : formatClock(elapsed)}
+
+ )
+}
+
+type VoiceRecordingTraceProps = ClockProps & {
+ getLevel: () => number
+ height: number
+ /** Names the trace for screen readers where no surrounding group does. */
+ levelLabel?: string
+}
+
+/**
+ * What every recording surface shows between its buttons: the trace of what
+ * the microphone heard, then the clock. Shared by the composer's recording bar
+ * and the settings page's transcription test, so the two read as one feature.
+ */
+export function VoiceRecordingTrace({ getLevel, height, levelLabel, ...clock }: VoiceRecordingTraceProps) {
+ return (
+
+
+
+
+
+
+ )
+}
+
+type VoiceRecordingBarProps = {
+ dictation: ComposerDictation
+ /** 44px touch targets, matching the composer's other mobile controls. */
+ mobile?: boolean
+}
+
+// Keep the caret in the composer: a click would otherwise blur it, and the
+// write-back position is the caret the user left there.
+const keepCaret = (event: MouseEvent) => event.preventDefault()
+
+/**
+ * The composer's toolbar while dictation is under way: cancel, what the
+ * microphone hears, the clock, stop (text goes into the draft) and send (text
+ * goes into the draft, then the draft is sent).
+ *
+ * The composer renders it in place of its toolbar controls for every phase but
+ * `idle`, keeping those controls mounted and hidden so their own state (an open
+ * model menu, a fetched context size) survives the recording.
+ */
+export function VoiceRecordingBar({ dictation, mobile = false }: VoiceRecordingBarProps) {
+ const t = useTranslation()
+ const { phase, startedAt, limitSeconds, sendRequested, getLevel, focusComposer } = dictation
+ const barRef = useRef(null)
+ const stopRef = useRef(null)
+ /** Focus entered the bar: the user is driving it from the keyboard. */
+ const focusedRef = useRef(false)
+ const starting = phase === 'starting'
+ const transcribing = phase === 'transcribing'
+
+ // Started from the keyboard, the microphone button that had focus is now
+ // hidden; hand focus to the control that ends the recording. Ending it hides
+ // the bar in turn, so focus goes back to the draft the text was written into.
+ // Mouse clicks never move focus here (every button prevents its mousedown),
+ // so a mouse user's focus is left wherever it was.
+ useLayoutEffect(() => {
+ if (document.activeElement?.closest('[hidden]')) stopRef.current?.focus()
+ const bar = barRef.current
+ return () => {
+ if (!focusedRef.current) return
+ // A disabled button drops focus to the body, so that counts as ours too.
+ const current = document.activeElement
+ if (!current || current === document.body || bar?.contains(current)) focusComposer()
+ }
+ }, [focusComposer])
+
+ if (phase === 'idle') return null
+
+ const iconSize = mobile ? 20 : 16
+ const size = mobile ? '2xl' : 'md'
+ const stopLabel = starting
+ ? t('voice.composer.starting')
+ : transcribing
+ ? t('voice.composer.transcribing')
+ : t('voice.composer.stop')
+
+ return (
+
{ focusedRef.current = true }}
+ className={`animate-overlay-in flex min-w-0 flex-1 items-center ${mobile ? 'gap-1' : 'gap-2'}`}
+ >
+ }
+ label={t('voice.composer.cancel')}
+ size={size}
+ shape="circle"
+ soft
+ onMouseDown={keepCaret}
+ onClick={dictation.cancel}
+ />
+
+ }
+ label={stopLabel}
+ size={size}
+ shape="circle"
+ soft
+ // The spinner belongs to whichever button the user is waiting on.
+ loading={transcribing && !sendRequested}
+ disabled={transcribing}
+ aria-busy={starting || transcribing ? true : undefined}
+ onMouseDown={keepCaret}
+ onClick={dictation.toggle}
+ />
+ {/* The composer's send key, so it reads as "send" without a word next to
+ it; it only differs in waiting for the text first. */}
+ }
+ />
+
+ )
+}
diff --git a/desktop/src/features/voiceInput/VoiceTrail.test.tsx b/desktop/src/features/voiceInput/VoiceTrail.test.tsx
new file mode 100644
index 00000000..16b4c2d7
--- /dev/null
+++ b/desktop/src/features/voiceInput/VoiceTrail.test.tsx
@@ -0,0 +1,235 @@
+import { act, cleanup, render } from '@testing-library/react'
+import '@testing-library/jest-dom'
+import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
+import { VoiceTrail } from './VoiceTrail'
+
+const WIDTH = 300
+const HEIGHT = 32
+const DOT = 2.5
+
+type Mark = { x: number; height: number; alpha: number; color: string }
+
+type FakeContext = {
+ setTransform: ReturnType
+ clearRect: ReturnType
+ beginPath: ReturnType
+ roundRect?: ReturnType
+ fill: ReturnType
+ fillRect: ReturnType
+ globalAlpha: number
+ fillStyle: string
+}
+
+let context: FakeContext
+let marks: Mark[]
+let frames: Map
+let nextFrameId: number
+let clock: number
+let disconnect: ReturnType
+let reducedMotion: boolean
+let hidden: boolean
+
+function makeContext(): FakeContext {
+ const record = (x: number, _y: number, width: number, height: number) => {
+ marks.push({ x: x + width / 2, height, alpha: ctx.globalAlpha, color: ctx.fillStyle })
+ }
+ const ctx: FakeContext = {
+ setTransform: vi.fn(),
+ clearRect: vi.fn(() => { marks = [] }),
+ beginPath: vi.fn(),
+ roundRect: vi.fn(record),
+ fill: vi.fn(),
+ fillRect: vi.fn(record),
+ globalAlpha: 1,
+ fillStyle: '',
+ }
+ return ctx
+}
+
+/** Runs queued frames for `ms` of animation time, one frame every 16 ms. */
+function play(ms: number) {
+ const end = clock + ms
+ while (clock < end) {
+ clock += 16
+ const batch = [...frames.values()]
+ frames.clear()
+ act(() => { for (const callback of batch) callback(clock) })
+ }
+}
+
+const bars = () => marks.filter((mark) => mark.height > DOT)
+const newest = () => marks.reduce((right, mark) => (mark.x > right.x ? mark : right))
+
+beforeEach(() => {
+ marks = []
+ context = makeContext()
+ frames = new Map()
+ nextFrameId = 1
+ clock = 1_000
+ disconnect = vi.fn()
+ reducedMotion = false
+ hidden = false
+
+ vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockImplementation(() => context as never)
+ Object.defineProperty(HTMLCanvasElement.prototype, 'clientWidth', { configurable: true, get: () => WIDTH })
+ vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => {
+ const id = nextFrameId++
+ frames.set(id, callback)
+ return id
+ })
+ vi.stubGlobal('cancelAnimationFrame', vi.fn((id: number) => { frames.delete(id) }))
+ vi.stubGlobal('ResizeObserver', class {
+ observe() {}
+ unobserve() {}
+ disconnect = disconnect
+ })
+ vi.stubGlobal('matchMedia', (query: string) => ({
+ get matches() { return query.includes('prefers-reduced-motion') && reducedMotion },
+ }))
+ Object.defineProperty(document, 'hidden', { configurable: true, get: () => hidden })
+ vi.stubGlobal('devicePixelRatio', 2)
+})
+
+afterEach(() => {
+ cleanup()
+ Reflect.deleteProperty(HTMLCanvasElement.prototype, 'clientWidth')
+ Reflect.deleteProperty(document, 'hidden')
+ vi.unstubAllGlobals()
+ vi.restoreAllMocks()
+})
+
+describe('VoiceTrail', () => {
+ it('marks the canvas decorative and sizes it at the device pixel ratio', () => {
+ const { container } = render( 0} frozen={false} height={HEIGHT} />)
+ const canvas = container.querySelector('canvas')!
+
+ expect(canvas).toHaveAttribute('aria-hidden', 'true')
+ expect(canvas.style.height).toBe(`${HEIGHT}px`)
+ expect(canvas.width).toBe(WIDTH * 2)
+ expect(canvas.height).toBe(HEIGHT * 2)
+ expect(context.setTransform).toHaveBeenLastCalledWith(2, 0, 0, 2, 0, 0)
+ })
+
+ it('fills the strip with faint dots before anything has been heard', () => {
+ render( 0} frozen={false} height={HEIGHT} />)
+
+ expect(marks.length).toBeGreaterThan(50)
+ expect(bars()).toEqual([])
+ // Only the newest slot has been listened to; the rest is not recorded yet.
+ const unrecorded = marks.filter((mark) => mark !== newest())
+ expect(Math.max(...unrecorded.map((mark) => mark.alpha))).toBeLessThanOrEqual(0.26)
+ })
+
+ it('turns speech into bars at the newest end, and keeps quiet as dots', () => {
+ let level = 0.8
+ render( level} frozen={false} height={HEIGHT} />)
+ play(300)
+
+ expect(bars().length).toBeGreaterThanOrEqual(3)
+ expect(newest().height).toBeGreaterThan(20)
+ expect(Math.max(...marks.map((mark) => mark.height))).toBeLessThanOrEqual(HEIGHT)
+
+ level = 0
+ play(600)
+ // The pause shows up as dots at the right end; the speech moved left.
+ expect(newest().height).toBe(DOT)
+ expect(bars().length).toBeGreaterThanOrEqual(3)
+ })
+
+ it('scrolls older marks to the left and fades them', () => {
+ let level = 0.8
+ render( level} frozen={false} height={HEIGHT} />)
+ play(150)
+ level = 0
+ // Long enough for the level to fall back and its last window to close.
+ play(500)
+ const before = bars()
+ const rightmost = Math.max(...before.map((mark) => mark.x))
+ const brightest = Math.max(...before.map((mark) => mark.alpha))
+
+ play(1_500)
+ const after = bars()
+ expect(after.length).toBe(before.length)
+ // 1.5 s at one mark per 75 ms is 20 marks of 5 px.
+ expect(rightmost - Math.max(...after.map((mark) => mark.x))).toBeCloseTo(100, 0)
+ expect(Math.max(...after.map((mark) => mark.alpha))).toBeLessThan(brightest)
+ })
+
+ it('stops listening when frozen, but keeps the trace where it was', () => {
+ const getLevel = vi.fn(() => 0.8)
+ const { rerender } = render()
+ play(300)
+ const recorded = bars().map((mark) => mark.x)
+
+ rerender()
+ getLevel.mockClear()
+ play(500)
+
+ expect(getLevel).not.toHaveBeenCalled()
+ expect(bars().map((mark) => mark.x)).toEqual(recorded)
+ })
+
+ it('sweeps a highlight across the frozen trace', () => {
+ const { rerender } = render( 0.8} frozen={false} height={HEIGHT} />)
+ play(1_000)
+ rerender( 0.8} frozen height={HEIGHT} />)
+
+ const alphaOverTime = new Set()
+ for (let step = 0; step < 10; step += 1) {
+ play(100)
+ alphaOverTime.add(Number(newest().alpha.toFixed(3)))
+ }
+ expect(alphaOverTime.size).toBeGreaterThan(3)
+ })
+
+ it('paints a frozen trace once and schedules nothing under reduced motion', () => {
+ reducedMotion = true
+ const { rerender } = render( 0.8} frozen={false} height={HEIGHT} />)
+ play(300)
+ expect(frames.size).toBe(1)
+
+ rerender( 0.8} frozen height={HEIGHT} />)
+
+ expect(frames.size).toBe(0)
+ expect(bars().length).toBeGreaterThanOrEqual(3)
+ })
+
+ it('draws square marks where roundRect is missing (Safari before 16)', () => {
+ delete context.roundRect
+ render( 0.8} frozen={false} height={HEIGHT} />)
+ play(200)
+
+ expect(context.fillRect).toHaveBeenCalled()
+ expect(bars().length).toBeGreaterThan(0)
+ })
+
+ it('cancels the frame and disconnects the observer on unmount', () => {
+ const getLevel = vi.fn(() => 0.5)
+ const { unmount } = render()
+ unmount()
+ getLevel.mockClear()
+ play(100)
+
+ expect(frames.size).toBe(0)
+ expect(disconnect).toHaveBeenCalled()
+ expect(getLevel).not.toHaveBeenCalled()
+ })
+
+ it('does not spin while the window is hidden, and resumes when it returns', () => {
+ render( 0.5} frozen={false} height={HEIGHT} />)
+ hidden = true
+ act(() => { document.dispatchEvent(new Event('visibilitychange')) })
+ expect(frames.size).toBe(0)
+
+ hidden = false
+ act(() => { document.dispatchEvent(new Event('visibilitychange')) })
+ expect(frames.size).toBe(1)
+ })
+
+ it('does not throw or animate when the canvas has no 2D context', () => {
+ vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(null)
+ render( 0.5} frozen={false} height={HEIGHT} />)
+
+ expect(frames.size).toBe(0)
+ })
+})
diff --git a/desktop/src/features/voiceInput/VoiceTrail.tsx b/desktop/src/features/voiceInput/VoiceTrail.tsx
new file mode 100644
index 00000000..75453d8a
--- /dev/null
+++ b/desktop/src/features/voiceInput/VoiceTrail.tsx
@@ -0,0 +1,217 @@
+import { memo, useEffect, useRef } from 'react'
+
+/** Loudness folded into one mark: its peak over this many seconds. */
+const SAMPLE_SECONDS = 0.075
+/** Centre-to-centre distance between marks, and the width of each. */
+const SPACING = 5
+const MARK_WIDTH = 2.5
+/** Below this a mark stays a dot: the room is quiet. */
+const QUIET = 0.06
+/** The oldest mark on screen fades to this share of the newest one. */
+const OLDEST_ALPHA = 0.5
+/** Marks dissolve over this many pixels at the left edge instead of being cut. */
+const EDGE_FADE = 28
+/** The part of the strip nothing has been recorded into yet. */
+const UNRECORDED_ALPHA = 0.26
+/** Level smoothing: marks should rise with a syllable and drop between words. */
+const ATTACK_SECONDS = 0.025
+const RELEASE_SECONDS = 0.09
+/** Longest step folded into one frame, so a stalled tab does not lurch on return. */
+const MAX_FRAME_SECONDS = 0.1
+/** While recognition runs, a highlight crosses the frozen trace this often. */
+const SWEEP_MS = 1200
+const SWEEP_WIDTH = 34
+/** Far more marks than the widest composer shows; older ones have scrolled away. */
+const MAX_MARKS = 600
+
+type Props = {
+ /** Input loudness in 0..1. Read every frame from a ref, so a new function identity never restarts the loop. */
+ getLevel: () => number
+ /** Recognition is running: stop listening, keep what was recorded, and sweep across it. */
+ frozen: boolean
+ height: number
+ className?: string
+}
+
+type Trace = {
+ /** One peak per finished sample window, oldest first. */
+ marks: number[]
+ /** Peak of the window being filled; drawn as the newest mark. */
+ peak: number
+ /** Seconds into that window. Also how far the strip has scrolled toward the next mark. */
+ windowElapsed: number
+ level: number
+}
+
+function readColor(canvas: HTMLCanvasElement, token: string): string {
+ const style = getComputedStyle(canvas)
+ return style.getPropertyValue(token).trim() || style.color
+}
+
+/**
+ * What the microphone heard so far, as a strip that scrolls right to left:
+ * dots while it is quiet, bars while someone speaks, older marks fainter.
+ *
+ * The one level display in the app: the composer's recording bar and the
+ * settings page's transcription test both draw it.
+ *
+ * Drawn straight onto a canvas from requestAnimationFrame: loudness never
+ * touches React state, so the parent does not re-render 60 times a second. The
+ * recorded marks live in a ref, so the trace survives the switch from recording
+ * to recognition, which only freezes it.
+ */
+export const VoiceTrail = memo(function VoiceTrail({ getLevel, frozen, height, className }: Props) {
+ const canvasRef = useRef(null)
+ const getLevelRef = useRef(getLevel)
+ getLevelRef.current = getLevel
+ const traceRef = useRef({ marks: [], peak: 0, windowElapsed: 0, level: 0 })
+
+ useEffect(() => {
+ const canvas = canvasRef.current
+ const context = canvas?.getContext('2d') ?? null
+ if (!canvas || !context) return
+
+ const trace = traceRef.current
+ const reducedMotion = typeof window.matchMedia === 'function'
+ ? window.matchMedia('(prefers-reduced-motion: reduce)')
+ : null
+ let width = 0
+ let pixelRatio = 1
+ let voiceColor = readColor(canvas, '--color-text-secondary')
+ let quietColor = readColor(canvas, '--color-text-tertiary')
+ let frame: number | null = null
+ let lastTime = 0
+ let sweepOrigin: number | null = null
+
+ const drawMark = (x: number, markHeight: number) => {
+ const top = height / 2 - markHeight / 2
+ const left = x - MARK_WIDTH / 2
+ context.beginPath()
+ // Safari before 16 has no roundRect; a square end is all that is lost.
+ if (typeof context.roundRect === 'function') {
+ context.roundRect(left, top, MARK_WIDTH, markHeight, MARK_WIDTH / 2)
+ context.fill()
+ } else {
+ context.fillRect(left, top, MARK_WIDTH, markHeight)
+ }
+ }
+
+ const paint = (now: number) => {
+ if (width <= 0) return
+ context.setTransform(pixelRatio, 0, 0, pixelRatio, 0, 0)
+ context.clearRect(0, 0, width, height)
+
+ const tallest = height - 6
+ const count = Math.floor((width - 4) / SPACING) + 2
+ const progress = trace.windowElapsed / SAMPLE_SECONDS
+ let sweep: number | null = null
+ if (frozen && !reducedMotion?.matches) {
+ // Starts and ends fully off the strip, so it does not pop in at an edge.
+ const reach = SWEEP_WIDTH * 2
+ sweepOrigin ??= now
+ sweep = (((now - sweepOrigin) % SWEEP_MS) / SWEEP_MS) * (width + 2 * reach) - reach
+ }
+
+ for (let index = 0; index < count; index += 1) {
+ const x = width - 3 - (index + progress) * SPACING
+ if (x < 1) break
+ const edge = Math.min(1, x / EDGE_FADE)
+ const value = index === 0 ? trace.peak : trace.marks[trace.marks.length - index]
+ if (value === undefined) {
+ context.globalAlpha = UNRECORDED_ALPHA * edge
+ context.fillStyle = quietColor
+ drawMark(x, MARK_WIDTH)
+ continue
+ }
+ const voiced = value >= QUIET
+ const markHeight = voiced
+ ? Math.min(tallest, MARK_WIDTH + (tallest - MARK_WIDTH) * Math.pow(Math.min(1, value * 1.25), 0.8))
+ : MARK_WIDTH
+ let alpha = (1 - (1 - OLDEST_ALPHA) * (index / count)) * edge
+ if (frozen) {
+ alpha *= sweep === null ? 0.6 : 0.35 + 0.65 * Math.exp(-(((x - sweep) / SWEEP_WIDTH) ** 2))
+ }
+ context.globalAlpha = alpha
+ context.fillStyle = voiced ? voiceColor : quietColor
+ drawMark(x, markHeight)
+ }
+ context.globalAlpha = 1
+ }
+
+ const layout = (nextWidth: number) => {
+ width = Math.max(0, Math.round(nextWidth))
+ pixelRatio = window.devicePixelRatio || 1
+ canvas.width = Math.round(width * pixelRatio)
+ canvas.height = Math.round(height * pixelRatio)
+ voiceColor = readColor(canvas, '--color-text-secondary')
+ quietColor = readColor(canvas, '--color-text-tertiary')
+ paint(performance.now())
+ }
+
+ const listen = (elapsed: number) => {
+ const target = Math.max(0, Math.min(1, getLevelRef.current()))
+ const seconds = target > trace.level ? ATTACK_SECONDS : RELEASE_SECONDS
+ trace.level += (target - trace.level) * (1 - Math.exp(-elapsed / seconds))
+ trace.peak = Math.max(trace.peak, trace.level)
+ trace.windowElapsed += elapsed
+ while (trace.windowElapsed >= SAMPLE_SECONDS) {
+ trace.marks.push(trace.peak)
+ trace.peak = trace.level
+ trace.windowElapsed -= SAMPLE_SECONDS
+ }
+ if (trace.marks.length > MAX_MARKS) trace.marks.splice(0, trace.marks.length - MAX_MARKS)
+ }
+
+ const tick = (now: number) => {
+ const elapsed = lastTime ? Math.min(MAX_FRAME_SECONDS, (now - lastTime) / 1000) : 0
+ lastTime = now
+ if (!frozen) listen(elapsed)
+ paint(now)
+ frame = requestAnimationFrame(tick)
+ }
+ const start = () => {
+ if (frame !== null || document.hidden) return
+ lastTime = 0
+ frame = requestAnimationFrame(tick)
+ }
+ const stop = () => {
+ if (frame === null) return
+ cancelAnimationFrame(frame)
+ frame = null
+ }
+ // A hidden window would spin the loop for nothing; resume when it returns.
+ const onVisibilityChange = () => (document.hidden ? stop() : start())
+
+ const observer = typeof ResizeObserver === 'function'
+ ? new ResizeObserver((entries) => {
+ const entry = entries[entries.length - 1]
+ if (entry) layout(entry.contentRect.width)
+ })
+ : null
+ observer?.observe(canvas)
+ layout(canvas.clientWidth)
+
+ // A frozen trace with reduced motion is a still picture: the one paint
+ // above is all it needs.
+ const animate = !frozen || !reducedMotion?.matches
+ if (animate) {
+ document.addEventListener('visibilitychange', onVisibilityChange)
+ start()
+ }
+ return () => {
+ stop()
+ observer?.disconnect()
+ document.removeEventListener('visibilitychange', onVisibilityChange)
+ }
+ }, [frozen, height])
+
+ return (
+
+ )
+})
diff --git a/desktop/src/features/voiceInput/VoiceWave.test.tsx b/desktop/src/features/voiceInput/VoiceWave.test.tsx
deleted file mode 100644
index 72b5fe3b..00000000
--- a/desktop/src/features/voiceInput/VoiceWave.test.tsx
+++ /dev/null
@@ -1,302 +0,0 @@
-import { act, cleanup, render } from '@testing-library/react'
-import '@testing-library/jest-dom'
-import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
-import { VoiceWave } from './VoiceWave'
-
-const HEIGHT = 44
-const MIDDLE = HEIGHT / 2
-
-type FakeContext = {
- setTransform: ReturnType
- clearRect: ReturnType
- beginPath: ReturnType
- moveTo: ReturnType
- lineTo: ReturnType
- stroke: ReturnType
- globalAlpha: number
- lineWidth: number
- strokeStyle: string
- lineCap: string
- lineJoin: string
-}
-
-let context: FakeContext
-let frames: Map
-let nextFrameId: number
-let cancelFrame: ReturnType
-let disconnect: ReturnType
-let resize: (width: number) => void
-let reducedMotion: boolean
-let hidden: boolean
-
-function makeContext(): FakeContext {
- return {
- setTransform: vi.fn(), clearRect: vi.fn(), beginPath: vi.fn(), moveTo: vi.fn(), lineTo: vi.fn(), stroke: vi.fn(),
- globalAlpha: 1, lineWidth: 1, strokeStyle: '', lineCap: '', lineJoin: '',
- }
-}
-
-/** Runs the queued frames once, `now` milliseconds on the animation clock. */
-function runFrames(now: number) {
- const batch = [...frames.values()]
- frames.clear()
- act(() => { for (const callback of batch) callback(now) })
-}
-
-/** Largest distance any drawn point of the last stroked layer reached from the centre line. */
-function peakOffset(): number {
- return Math.max(...context.lineTo.mock.calls.map(([, y]) => Math.abs((y as number) - MIDDLE)))
-}
-
-/** Runs one frame and returns the y values the first (top) layer drew in it. */
-function topLayerAt(now: number): number[] {
- context.moveTo.mockClear()
- context.lineTo.mockClear()
- runFrames(now)
- const points = context.lineTo.mock.calls
- return points.slice(0, points.length / 3).map(([, y]) => y as number)
-}
-
-function expectClose(actual: number[], expected: number[]) {
- expect(actual).toHaveLength(expected.length)
- actual.forEach((y, index) => expect(y).toBeCloseTo(expected[index]!, 6))
-}
-
-beforeEach(() => {
- context = makeContext()
- frames = new Map()
- nextFrameId = 1
- cancelFrame = vi.fn((id: number) => { frames.delete(id) })
- disconnect = vi.fn()
- reducedMotion = false
- hidden = false
-
- vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockImplementation(() => context as never)
- Object.defineProperty(HTMLCanvasElement.prototype, 'clientWidth', { configurable: true, get: () => 300 })
- vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => {
- const id = nextFrameId++
- frames.set(id, callback)
- return id
- })
- vi.stubGlobal('cancelAnimationFrame', cancelFrame)
- vi.stubGlobal('ResizeObserver', class {
- constructor(callback: ResizeObserverCallback) {
- resize = (width) => callback([{ contentRect: { width } } as ResizeObserverEntry], this as unknown as ResizeObserver)
- }
- observe() {}
- unobserve() {}
- disconnect = disconnect
- })
- vi.stubGlobal('matchMedia', (query: string) => ({
- get matches() { return query.includes('prefers-reduced-motion') && reducedMotion },
- }))
- Object.defineProperty(document, 'hidden', { configurable: true, get: () => hidden })
- vi.stubGlobal('devicePixelRatio', 2)
-})
-
-afterEach(() => {
- cleanup()
- Reflect.deleteProperty(HTMLCanvasElement.prototype, 'clientWidth')
- Reflect.deleteProperty(document, 'hidden')
- vi.unstubAllGlobals()
- vi.restoreAllMocks()
-})
-
-describe('VoiceWave', () => {
- it('reads the level and draws three layers on every frame while active', () => {
- const getLevel = vi.fn(() => 0.5)
- const layers: Array<{ alpha: number; width: number }> = []
- context.stroke.mockImplementation(() => { layers.push({ alpha: context.globalAlpha, width: context.lineWidth }) })
- render()
- layers.length = 0
-
- expect(frames.size).toBe(1)
- runFrames(1_000)
- expect(getLevel).toHaveBeenCalledTimes(1)
- // The top layer is opaque and full width; the lower two are fainter and thinner.
- expect(layers).toEqual([
- { alpha: 1, width: 2 },
- { alpha: 0.42, width: 1.25 },
- { alpha: 0.26, width: 1 },
- ])
- expect(context.lineCap).toBe('round')
- runFrames(1_016)
- expect(getLevel).toHaveBeenCalledTimes(2)
- expect(layers).toHaveLength(6)
- })
-
- it('marks the canvas decorative and sizes it to the container at the device pixel ratio', () => {
- const { container } = render( 0} active />)
- const canvas = container.querySelector('canvas')!
-
- expect(canvas).toHaveAttribute('aria-hidden', 'true')
- expect(canvas.style.height).toBe('44px')
- // 300 css px wide, 44 tall, at 2x.
- expect(canvas.width).toBe(600)
- expect(canvas.height).toBe(88)
- expect(context.setTransform).toHaveBeenLastCalledWith(2, 0, 0, 2, 0, 0)
-
- act(() => resize(420))
- expect(canvas.width).toBe(840)
- expect(canvas.height).toBe(88)
- })
-
- it('pins both ends to the centre line so the wave floats in the middle', () => {
- render( 1} active />)
- runFrames(1_000)
- runFrames(1_500)
-
- const first = context.moveTo.mock.calls[0]!
- expect(first[0]).toBe(0)
- expect(first[1]).toBeCloseTo(MIDDLE, 5)
- const last = context.lineTo.mock.calls.filter(([x]) => x === 300)[0]!
- expect(last[1]).toBeCloseTo(MIDDLE, 5)
- // ... while the middle of the wave does swing.
- expect(peakOffset()).toBeGreaterThan(8)
- })
-
- it('rises quickly on loud input and falls back slowly', () => {
- let level = 0
- render( level} active />)
- runFrames(1_000)
- context.lineTo.mockClear()
- runFrames(1_016)
- const idle = peakOffset()
-
- level = 1
- context.lineTo.mockClear()
- runFrames(1_032)
- const attacked = peakOffset()
- // One 16 ms frame already covers a good part of the way up.
- expect(attacked).toBeGreaterThan(idle + 3)
-
- for (let time = 1_048; time < 1_300; time += 16) runFrames(time)
- context.lineTo.mockClear()
- runFrames(1_316)
- const loud = peakOffset()
-
- level = 0
- context.lineTo.mockClear()
- runFrames(1_332)
- const released = peakOffset()
- // Still most of the way up one frame after the input went quiet ...
- expect(released).toBeGreaterThan(loud * 0.8)
- // ... and settled back down a couple of seconds later.
- for (let time = 1_348; time < 3_400; time += 16) runFrames(time)
- context.lineTo.mockClear()
- runFrames(3_416)
- expect(peakOffset()).toBeLessThan(idle + 1)
- })
-
- it('keeps a near-flat line when there is no sound', () => {
- render( 0} active />)
- runFrames(1_000)
- runFrames(1_016)
-
- expect(peakOffset()).toBeGreaterThan(0.5)
- expect(peakOffset()).toBeLessThan(3)
- })
-
- it('draws one calm frame and schedules nothing while inactive', () => {
- const getLevel = vi.fn(() => 1)
- render()
-
- expect(frames.size).toBe(0)
- expect(getLevel).not.toHaveBeenCalled()
- // A single static paint at mount: three layers, nothing more.
- expect(context.stroke).toHaveBeenCalledTimes(3)
- expect(peakOffset()).toBeLessThan(3)
- })
-
- it('stops reading the level and cancels the frame when it goes inactive', () => {
- const getLevel = vi.fn(() => 0.5)
- const { rerender } = render()
- runFrames(1_000)
- const readsWhileActive = getLevel.mock.calls.length
-
- rerender()
-
- expect(cancelFrame).toHaveBeenCalled()
- expect(frames.size).toBe(0)
- runFrames(1_016)
- expect(getLevel).toHaveBeenCalledTimes(readsWhileActive)
- })
-
- it('cancels the frame, disconnects the observer and stops reading on unmount', () => {
- const getLevel = vi.fn(() => 0.5)
- const { unmount } = render()
- runFrames(1_000)
- const reads = getLevel.mock.calls.length
-
- unmount()
-
- expect(cancelFrame).toHaveBeenCalled()
- expect(disconnect).toHaveBeenCalled()
- expect(frames.size).toBe(0)
- expect(getLevel).toHaveBeenCalledTimes(reads)
- })
-
- it('does not restart the loop when the parent passes a new getLevel function', () => {
- const first = vi.fn(() => 0.2)
- const second = vi.fn(() => 0.8)
- const { rerender } = render()
- runFrames(1_000)
-
- rerender()
-
- expect(cancelFrame).not.toHaveBeenCalled()
- runFrames(1_016)
- expect(second).toHaveBeenCalled()
- expect(first).toHaveBeenCalledTimes(1)
- })
-
- it('does not throw or animate when the canvas has no 2D context', () => {
- vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(null)
-
- expect(() => render( 1} active />)).not.toThrow()
- expect(frames.size).toBe(0)
- })
-
- it('does not spin while the window is hidden, and resumes when it returns', () => {
- hidden = true
- const getLevel = vi.fn(() => 0.5)
- render()
- expect(frames.size).toBe(0)
-
- hidden = false
- act(() => { document.dispatchEvent(new Event('visibilitychange')) })
- expect(frames.size).toBe(1)
- runFrames(1_000)
- expect(getLevel).toHaveBeenCalledTimes(1)
-
- hidden = true
- act(() => { document.dispatchEvent(new Event('visibilitychange')) })
- expect(frames.size).toBe(0)
- expect(cancelFrame).toHaveBeenCalled()
- })
-
- it('flows forward over time normally', () => {
- render( 0.6} active />)
- runFrames(1_000)
- const before = topLayerAt(1_100)
- const after = topLayerAt(1_200)
-
- expect(after).not.toEqual(before)
- })
-
- it('holds the phase under prefers-reduced-motion and only lets the level change the height', () => {
- reducedMotion = true
- let level = 0.6
- render( level} active />)
- // Let the smoothed level settle so only the phase could differ between frames.
- for (let time = 1_000; time < 3_000; time += 16) runFrames(time)
- const before = topLayerAt(3_016)
- const after = topLayerAt(3_500)
- expectClose(after, before)
-
- level = 1
- for (let time = 3_516; time < 5_000; time += 16) runFrames(time)
- const louder = topLayerAt(5_016)
- expect(Math.max(...louder.map(y => Math.abs(y - MIDDLE)))).toBeGreaterThan(Math.max(...after.map(y => Math.abs(y - MIDDLE))))
- })
-})
diff --git a/desktop/src/features/voiceInput/VoiceWave.tsx b/desktop/src/features/voiceInput/VoiceWave.tsx
deleted file mode 100644
index 6ae96c82..00000000
--- a/desktop/src/features/voiceInput/VoiceWave.tsx
+++ /dev/null
@@ -1,153 +0,0 @@
-import { memo, useEffect, useRef } from 'react'
-
-const HEIGHT = 44
-/** Idle undulation: the line stays visibly alive, but reads as flat. */
-const BASE_AMPLITUDE = 0.04 * HEIGHT
-/** Level smoothing time constants: rise quickly on speech, fall back slowly. */
-const ATTACK_SECONDS = 0.06
-const RELEASE_SECONDS = 0.35
-/** Longest step folded into one frame, so a stalled tab does not lurch on return. */
-const MAX_FRAME_SECONDS = 0.1
-const POINT_SPACING = 2
-
-/**
- * Three overlapping sine waves. Only the top one is fully opaque; the others
- * are thinner and fainter so the shape reads as one motion, not three lines.
- * `speed` is radians per second; all flow the same way at slightly different
- * rates, which keeps the layers drifting against each other.
- */
-const LAYERS = [
- { wavelength: 150, speed: 1.8, offset: 0, scale: 1, lineWidth: 2, alpha: 1 },
- { wavelength: 104, speed: 1.3, offset: 1.7, scale: 0.68, lineWidth: 1.25, alpha: 0.42 },
- { wavelength: 76, speed: 2.4, offset: 3.4, scale: 0.48, lineWidth: 1, alpha: 0.26 },
-] as const
-
-type Props = {
- /** Input loudness in 0..1. Read every frame from a ref, so a new function identity never restarts the loop. */
- getLevel: () => number
- /** Animate from `getLevel`. Inactive draws a single calm line and does no work. */
- active: boolean
- className?: string
-}
-
-function readColor(canvas: HTMLCanvasElement): string {
- const style = getComputedStyle(canvas)
- return style.getPropertyValue('--color-brand').trim() || style.color
-}
-
-/**
- * A quiet, layered waveform for the recording level.
- *
- * Drawn straight onto a canvas from requestAnimationFrame: level changes never
- * touch React state, so the parent does not re-render 60 times a second.
- */
-export const VoiceWave = memo(function VoiceWave({ getLevel, active, className }: Props) {
- const canvasRef = useRef(null)
- const getLevelRef = useRef(getLevel)
- getLevelRef.current = getLevel
-
- useEffect(() => {
- const canvas = canvasRef.current
- const context = canvas?.getContext('2d') ?? null
- if (!canvas || !context) return
-
- const reducedMotion = typeof window.matchMedia === 'function'
- ? window.matchMedia('(prefers-reduced-motion: reduce)')
- : null
- let width = 0
- let pixelRatio = 1
- let color = readColor(canvas)
- let frame: number | null = null
- let lastTime = 0
- let phase = 0
- let level = 0
-
- const paint = () => {
- if (width <= 0) return
- context.setTransform(pixelRatio, 0, 0, pixelRatio, 0, 0)
- context.clearRect(0, 0, width, HEIGHT)
- context.lineCap = 'round'
- context.lineJoin = 'round'
- context.strokeStyle = color
-
- const middle = HEIGHT / 2
- const amplitude = BASE_AMPLITUDE + level * (middle - BASE_AMPLITUDE - 2)
- for (const layer of LAYERS) {
- context.globalAlpha = layer.alpha
- context.lineWidth = layer.lineWidth
- context.beginPath()
- for (let x = 0; x <= width; x += POINT_SPACING) {
- // The envelope pins both ends to the centre line, so the wave floats
- // in the middle instead of hitting the edges.
- const envelope = Math.pow(Math.sin((Math.PI * x) / width), 1.6)
- const angle = (x / layer.wavelength) * Math.PI * 2 - phase * layer.speed + layer.offset
- const y = middle + Math.sin(angle) * amplitude * layer.scale * envelope
- if (x === 0) context.moveTo(x, y)
- else context.lineTo(x, y)
- }
- context.stroke()
- }
- context.globalAlpha = 1
- }
-
- const layout = (nextWidth: number) => {
- width = Math.max(0, Math.round(nextWidth))
- pixelRatio = window.devicePixelRatio || 1
- canvas.width = Math.round(width * pixelRatio)
- canvas.height = Math.round(HEIGHT * pixelRatio)
- color = readColor(canvas)
- paint()
- }
-
- const tick = (now: number) => {
- const elapsed = lastTime ? Math.min(MAX_FRAME_SECONDS, (now - lastTime) / 1000) : 0
- lastTime = now
- const target = Math.max(0, Math.min(1, getLevelRef.current()))
- const seconds = target > level ? ATTACK_SECONDS : RELEASE_SECONDS
- level += (target - level) * (1 - Math.exp(-elapsed / seconds))
- if (!reducedMotion?.matches) phase += elapsed
- paint()
- frame = requestAnimationFrame(tick)
- }
- const start = () => {
- if (frame !== null || document.hidden) return
- lastTime = 0
- frame = requestAnimationFrame(tick)
- }
- const stop = () => {
- if (frame === null) return
- cancelAnimationFrame(frame)
- frame = null
- }
- // A hidden window would spin the loop for nothing; resume when it returns.
- const onVisibilityChange = () => (document.hidden ? stop() : start())
-
- const observer = typeof ResizeObserver === 'function'
- ? new ResizeObserver((entries) => {
- const entry = entries[entries.length - 1]
- if (entry) layout(entry.contentRect.width)
- })
- : null
- observer?.observe(canvas)
- layout(canvas.clientWidth)
-
- if (active) {
- document.addEventListener('visibilitychange', onVisibilityChange)
- start()
- }
- return () => {
- stop()
- observer?.disconnect()
- document.removeEventListener('visibilitychange', onVisibilityChange)
- }
- }, [active])
-
- return (
-
- )
-})
diff --git a/desktop/src/features/voiceInput/useComposerDictation.ts b/desktop/src/features/voiceInput/useComposerDictation.ts
index 53f28a16..85d46829 100644
--- a/desktop/src/features/voiceInput/useComposerDictation.ts
+++ b/desktop/src/features/voiceInput/useComposerDictation.ts
@@ -39,6 +39,8 @@ type Run = {
revisionAtStart: number
point: InsertionPoint
stopping: boolean
+ /** Send the draft once the text is written. Can be asked for until recognition ends. */
+ send: boolean
}
type ComposerDictationOptions = {
@@ -49,6 +51,11 @@ type ComposerDictationOptions = {
blocked: boolean
/** Identifies what the composer is editing; a change abandons any dictation. */
contextKey: string | null | undefined
+ /**
+ * Sends the draft through the composer's own submit path. Called after the
+ * render that carries the dictated text, so it reads the draft with it.
+ */
+ onSubmit?: () => void
}
/** Recordings shorter than this are a mis-tap, not speech. */
@@ -90,7 +97,7 @@ function issueFromError(error: unknown): DictationIssue {
* Shared by both composers (ChatInput and EmptySession) so the write-back rules
* cannot drift between them.
*/
-export function useComposerDictation({ composerRef, draft, blocked, contextKey }: ComposerDictationOptions) {
+export function useComposerDictation({ composerRef, draft, blocked, contextKey, onSubmit }: ComposerDictationOptions) {
const providerId = useVoiceInputStore(state => state.catalog?.preferences.providerId)
const language = useVoiceInputStore(state => state.catalog?.preferences.language)
const maxSeconds = useVoiceInputStore(state => state.catalog?.limits.maxAudioSeconds)
@@ -99,6 +106,8 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
const [issue, setIssue] = useState(null)
const [pendingText, setPendingText] = useState(null)
const [startedAt, setStartedAt] = useState(0)
+ const [limitSeconds, setLimitSeconds] = useState(0)
+ const [sendRequested, setSendRequested] = useState(false)
const activeRef = useRef(null)
const draftRef = useRef(draft)
@@ -108,15 +117,24 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
const composingRef = useRef(false)
const pendingRef = useRef(null)
const previousContextRef = useRef(contextKey)
+ const onSubmitRef = useRef(onSubmit)
+ /** Dictated text was written for a send; submit once the draft carries it. */
+ const submitAfterWriteRef = useRef(false)
blockedRef.current = blocked
settingsRef.current = { providerId, language, maxSeconds }
+ onSubmitRef.current = onSubmit
useEffect(() => {
if (draftRef.current !== draft) {
draftRef.current = draft
revisionRef.current += 1
}
+ // Submitting from `deliver` itself would send the draft as the parent last
+ // rendered it — without the text that was just written.
+ if (!submitAfterWriteRef.current) return
+ submitAfterWriteRef.current = false
+ onSubmitRef.current?.()
}, [draft])
const setPending = useCallback((text: string | null) => {
@@ -134,15 +152,26 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
return { start: length, end: length }
}, [composerRef])
- const writeText = useCallback((text: string, point: InsertionPoint) => {
+ const writeText = useCallback((text: string, point: InsertionPoint): boolean => {
const composer = composerRef.current
- if (!composer) return
+ if (!composer) return false
const current = draftRef.current
const start = Math.min(point.start, current.length)
const end = Math.min(Math.max(point.end, start), current.length)
- composer.insertTextAtOffsets(start, end, withDictationSpacing(current.slice(0, start), text, current.slice(end)))
+ composer.insertTextAtOffsets(
+ start,
+ end,
+ withDictationSpacing(current.slice(0, start), text, current.slice(end)),
+ { flash: true },
+ )
+ return true
}, [composerRef])
+ const settle = useCallback(() => {
+ setPhase('idle')
+ setSendRequested(false)
+ }, [])
+
const cancel = useCallback(() => {
const run = activeRef.current
activeRef.current = null
@@ -150,8 +179,9 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
run.controller.abort()
run.recording?.cancel()
}
- setPhase('idle')
- }, [])
+ submitAfterWriteRef.current = false
+ settle()
+ }, [settle])
const abandon = useCallback(() => {
cancel()
@@ -172,9 +202,11 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
composing: composingRef.current,
})
if (placement === 'insert') {
- writeText(text, run.point)
+ if (writeText(text, run.point) && run.send) submitAfterWriteRef.current = true
return
}
+ // Held text is never sent on its own: the draft changed, or cannot take
+ // text, so the user has to look at it first.
setPending(text)
}, [setPending, writeText])
@@ -184,7 +216,7 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
setPhase('transcribing')
const fail = (next: DictationIssue) => {
activeRef.current = null
- setPhase('idle')
+ settle()
setIssue(next)
}
try {
@@ -206,7 +238,7 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
})
if (activeRef.current !== run) return
activeRef.current = null
- setPhase('idle')
+ settle()
deliver(run, transcript.text)
} catch (error) {
if (activeRef.current !== run) return
@@ -216,7 +248,7 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
if (next === 'notReady') void useVoiceInputStore.getState().loadCatalog({ force: true })
fail(next)
}
- }, [deliver])
+ }, [deliver, settle])
const start = useCallback(async () => {
const settings = settingsRef.current
@@ -227,11 +259,14 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
revisionAtStart: revisionRef.current,
point: capturePoint(),
stopping: false,
+ send: false,
}
activeRef.current = run
+ submitAfterWriteRef.current = false
// Starting over is a deliberate choice; the previous held text goes.
setPending(null)
setIssue(null)
+ setSendRequested(false)
setPhase('starting')
let recording: ActiveRecording
@@ -244,14 +279,14 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
onInterrupted: (error) => {
if (activeRef.current !== run) return
activeRef.current = null
- setPhase('idle')
+ settle()
setIssue(issueFromRecorder(error.code))
},
})
} catch (error) {
if (activeRef.current !== run) return
activeRef.current = null
- setPhase('idle')
+ settle()
setIssue(issueFromError(error))
return
}
@@ -262,8 +297,9 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
}
run.recording = recording
setStartedAt(Date.now())
+ setLimitSeconds(settings.maxSeconds)
setPhase('recording')
- }, [capturePoint, finish, setPending])
+ }, [capturePoint, finish, setPending, settle])
const toggle = useCallback(() => {
const run = activeRef.current
@@ -276,6 +312,18 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
}
}, [cancel, finish, start])
+ /**
+ * Ends the recording and sends the draft once the text is in it. Also works
+ * while recognition is running, for a user who decides late.
+ */
+ const stopAndSend = useCallback(() => {
+ const run = activeRef.current
+ if (!run?.recording || run.send) return
+ run.send = true
+ setSendRequested(true)
+ void finish(run)
+ }, [finish])
+
const insertPending = useCallback(() => {
const text = pendingRef.current
if (text === null || blockedRef.current) return
@@ -288,6 +336,8 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
const dismissIssue = useCallback(() => setIssue(null), [])
+ const focusComposer = useCallback(() => composerRef.current?.focus(), [composerRef])
+
const onCompositionStart = useCallback(() => {
composingRef.current = true
}, [])
@@ -333,24 +383,32 @@ export function useComposerDictation({ composerRef, draft, blocked, contextKey }
issue,
pendingText,
startedAt,
+ limitSeconds,
+ sendRequested,
toggle,
+ stopAndSend,
cancel,
insertPending,
dismissPending,
dismissIssue,
getLevel,
+ focusComposer,
compositionHandlers: { onCompositionStart, onCompositionEnd },
}), [
phase,
issue,
pendingText,
startedAt,
+ limitSeconds,
+ sendRequested,
toggle,
+ stopAndSend,
cancel,
insertPending,
dismissPending,
dismissIssue,
getLevel,
+ focusComposer,
onCompositionStart,
onCompositionEnd,
])
diff --git a/desktop/src/i18n/locales/en.ts b/desktop/src/i18n/locales/en.ts
index 91d0acdf..87bafeef 100644
--- a/desktop/src/i18n/locales/en.ts
+++ b/desktop/src/i18n/locales/en.ts
@@ -4071,10 +4071,17 @@ Row 9, all 8 cells: continuing from straight down, turning left through lower-le
'voice.settings.test.stats': 'Audio {audio} s · Inference {inference} s',
'voice.settings.test.playback': 'Play back the recording',
'voice.composer.start': 'Dictate',
+ 'voice.composer.needsModel': 'Dictate — download the speech model in Settings first',
+ 'voice.composer.modelDownloading': 'Speech model downloading — open Settings to see progress',
'voice.composer.starting': 'Opening microphone… (click to cancel)',
'voice.composer.stop': 'Stop recording and transcribe',
'voice.composer.transcribing': 'Transcribing…',
'voice.composer.recordingHint': 'Recording — press Esc to cancel',
+ 'voice.composer.recordingBar': 'Voice recording',
+ 'voice.composer.cancel': 'Cancel recording (Esc)',
+ 'voice.composer.sendNow': 'Transcribe and send',
+ 'voice.composer.transcribingThenSend': 'Transcribing, then sending…',
+ 'voice.composer.remaining': '{time} left',
'voice.composer.insertText': 'Insert text',
'voice.composer.discard': 'Discard',
'voice.composer.dismiss': 'Dismiss',
diff --git a/desktop/src/i18n/locales/jp.ts b/desktop/src/i18n/locales/jp.ts
index d8920f9a..acd1c22e 100644
--- a/desktop/src/i18n/locales/jp.ts
+++ b/desktop/src/i18n/locales/jp.ts
@@ -4072,10 +4072,17 @@ export const jp: Record = {
'voice.settings.test.stats': '音声 {audio} 秒 · 推論 {inference} 秒',
'voice.settings.test.playback': '録音を再生',
'voice.composer.start': '音声入力',
+ 'voice.composer.needsModel': '音声入力:先に設定で音声モデルをダウンロード',
+ 'voice.composer.modelDownloading': '音声モデルをダウンロード中 · クリックで進行状況を表示',
'voice.composer.starting': 'マイクを開いています…(クリックでキャンセル)',
'voice.composer.stop': '録音を停止して認識',
'voice.composer.transcribing': '認識中…',
'voice.composer.recordingHint': '録音中 · Esc でキャンセル',
+ 'voice.composer.recordingBar': '音声録音',
+ 'voice.composer.cancel': '録音をキャンセル(Esc)',
+ 'voice.composer.sendNow': '認識してそのまま送信',
+ 'voice.composer.transcribingThenSend': '認識後に送信…',
+ 'voice.composer.remaining': '残り {time}',
'voice.composer.insertText': 'テキストを挿入',
'voice.composer.discard': '破棄',
'voice.composer.dismiss': '閉じる',
diff --git a/desktop/src/i18n/locales/kr.ts b/desktop/src/i18n/locales/kr.ts
index d6e9e8de..7ed0c28d 100644
--- a/desktop/src/i18n/locales/kr.ts
+++ b/desktop/src/i18n/locales/kr.ts
@@ -4074,10 +4074,17 @@ export const kr: Record = {
'voice.settings.test.stats': '오디오 {audio}초 · 추론 {inference}초',
'voice.settings.test.playback': '녹음 재생',
'voice.composer.start': '음성 입력',
+ 'voice.composer.needsModel': '음성 입력: 먼저 설정에서 음성 모델을 다운로드하세요',
+ 'voice.composer.modelDownloading': '음성 모델 다운로드 중 · 클릭하면 진행 상황 보기',
'voice.composer.starting': '마이크를 여는 중… (클릭하면 취소)',
'voice.composer.stop': '녹음을 멈추고 인식',
'voice.composer.transcribing': '인식 중…',
'voice.composer.recordingHint': '녹음 중 · Esc로 취소',
+ 'voice.composer.recordingBar': '음성 녹음',
+ 'voice.composer.cancel': '녹음 취소 (Esc)',
+ 'voice.composer.sendNow': '인식 후 바로 보내기',
+ 'voice.composer.transcribingThenSend': '인식 후 보내는 중…',
+ 'voice.composer.remaining': '{time} 남음',
'voice.composer.insertText': '텍스트 삽입',
'voice.composer.discard': '버리기',
'voice.composer.dismiss': '닫기',
diff --git a/desktop/src/i18n/locales/zh-TW.ts b/desktop/src/i18n/locales/zh-TW.ts
index 7b54112e..9dcc3e8d 100644
--- a/desktop/src/i18n/locales/zh-TW.ts
+++ b/desktop/src/i18n/locales/zh-TW.ts
@@ -4071,10 +4071,17 @@ export const zh: Record = {
'voice.settings.test.stats': '音訊 {audio} 秒 · 推論 {inference} 秒',
'voice.settings.test.playback': '重播錄音',
'voice.composer.start': '語音輸入',
+ 'voice.composer.needsModel': '語音輸入:先到設定裡下載語音模型',
+ 'voice.composer.modelDownloading': '語音模型下載中,點擊查看進度',
'voice.composer.starting': '正在開啟麥克風…(點擊取消)',
'voice.composer.stop': '停止錄音並辨識',
'voice.composer.transcribing': '辨識中…',
'voice.composer.recordingHint': '錄音中,按 Esc 取消',
+ 'voice.composer.recordingBar': '語音錄音',
+ 'voice.composer.cancel': '取消錄音(Esc)',
+ 'voice.composer.sendNow': '辨識後直接傳送',
+ 'voice.composer.transcribingThenSend': '辨識後傳送…',
+ 'voice.composer.remaining': '還剩 {time}',
'voice.composer.insertText': '插入文字',
'voice.composer.discard': '捨棄',
'voice.composer.dismiss': '關閉提示',
diff --git a/desktop/src/i18n/locales/zh.ts b/desktop/src/i18n/locales/zh.ts
index 9591e49e..db1ba831 100644
--- a/desktop/src/i18n/locales/zh.ts
+++ b/desktop/src/i18n/locales/zh.ts
@@ -4070,10 +4070,17 @@ export const zh: Record = {
'voice.settings.test.stats': '音频 {audio} 秒 · 推理 {inference} 秒',
'voice.settings.test.playback': '回放录音',
'voice.composer.start': '语音输入',
+ 'voice.composer.needsModel': '语音输入:先到设置里下载语音模型',
+ 'voice.composer.modelDownloading': '语音模型下载中,点击查看进度',
'voice.composer.starting': '正在打开麦克风…(点击取消)',
'voice.composer.stop': '停止录音并识别',
'voice.composer.transcribing': '识别中…',
'voice.composer.recordingHint': '录音中,按 Esc 取消',
+ 'voice.composer.recordingBar': '语音录音',
+ 'voice.composer.cancel': '取消录音(Esc)',
+ 'voice.composer.sendNow': '识别后直接发送',
+ 'voice.composer.transcribingThenSend': '识别后发送…',
+ 'voice.composer.remaining': '还剩 {time}',
'voice.composer.insertText': '插入文字',
'voice.composer.discard': '丢弃',
'voice.composer.dismiss': '关闭提示',
diff --git a/desktop/src/pages/EmptySession.test.tsx b/desktop/src/pages/EmptySession.test.tsx
index b893bf33..e8a1c37c 100644
--- a/desktop/src/pages/EmptySession.test.tsx
+++ b/desktop/src/pages/EmptySession.test.tsx
@@ -1554,6 +1554,10 @@ describe('EmptySession', () => {
mocks.voiceTranscribe.mockImplementation(() => new Promise((resolve) => {
finishTranscription = (text) => resolve({ text, audioSeconds: 2, inferenceSeconds: 0.1 })
}))
+ // Voice input exists only in the desktop app (the outer beforeEach resets this).
+ mocks.isTauriRuntime = true
+ // jsdom has no canvas; the recording bar's trace draws nothing without one.
+ vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(null)
})
async function dictate() {
@@ -1582,6 +1586,12 @@ describe('EmptySession', () => {
expect(screen.queryByTestId('voice-input')).toBeNull()
})
+ it('does not render the microphone in the browser (H5)', () => {
+ mocks.isTauriRuntime = false
+ render()
+ expect(screen.queryByTestId('voice-input')).toBeNull()
+ })
+
it('writes dictated text at the caret without starting a session', async () => {
render()
setComposerText('ab', 1)
@@ -1597,6 +1607,32 @@ describe('EmptySession', () => {
expect(mocks.createSession).not.toHaveBeenCalled()
})
+ it('starts the session with the dictated text when sent from the recording bar', async () => {
+ render()
+ await act(async () => {
+ fireEvent.click(screen.getByRole('button', { name: 'Dictate' }))
+ })
+ // The bar takes the toolbar's row: its controls, Run included, are hidden.
+ expect(screen.getByTestId('voice-recording-bar')).toBeVisible()
+ expect(screen.queryByRole('button', { name: 'Run' })).toBeNull()
+
+ await act(async () => {
+ fireEvent.click(await screen.findByRole('button', { name: 'Transcribe and send' }))
+ })
+ expect(mocks.createSession).not.toHaveBeenCalled()
+ await act(async () => {
+ finishTranscription('你好')
+ })
+
+ await waitFor(() => {
+ expect(mocks.wsSend).toHaveBeenCalledWith('draft-session', {
+ type: 'user_message',
+ content: '你好',
+ attachments: [],
+ })
+ })
+ })
+
it('keeps the text aside when the draft was edited while it was being recognised', async () => {
render()
setComposerText('hello', 5)
diff --git a/desktop/src/pages/EmptySession.tsx b/desktop/src/pages/EmptySession.tsx
index bde3fee6..6cd008d3 100644
--- a/desktop/src/pages/EmptySession.tsx
+++ b/desktop/src/pages/EmptySession.tsx
@@ -79,6 +79,7 @@ import type { PermissionMode } from '../types/settings'
import type { SlashCommandOption } from '../components/chat/composerUtils'
import { useComposerDictation } from '@/features/voiceInput/useComposerDictation'
import { VoiceInputButton } from '@/features/voiceInput/VoiceInputButton'
+import { VoiceRecordingBar } from '@/features/voiceInput/VoiceRecordingBar'
type Attachment = ComposerAttachment
@@ -204,7 +205,12 @@ export function EmptySession() {
draft: input,
blocked: isSubmitting,
contextKey: 'empty-session',
+ // Called from an effect, after this render has declared `handleSubmit`.
+ onSubmit: () => { void handleSubmit() },
})
+ // While dictating, the toolbar's controls stay mounted but hidden, and the
+ // recording bar takes their row.
+ const dictationLive = dictation.phase !== 'idle'
useEffect(() => {
composerRef.current?.focus()
@@ -910,7 +916,8 @@ export function EmptySession() {
-
+ {dictationLive && }
+
{/* Hand-rolled like ChatInput's: a quiet 28px square on
desktop, the 44px touch minimum on a phone. */}
@@ -973,7 +980,10 @@ export function EmptySession() {
)}
-
+
{
let frames: Array
let now: number
const canvasContext = {
- setTransform: vi.fn(), clearRect: vi.fn(), beginPath: vi.fn(), moveTo: vi.fn(), lineTo: vi.fn(), stroke: vi.fn(),
+ setTransform: vi.fn(), clearRect: vi.fn(), beginPath: vi.fn(), roundRect: vi.fn(), fill: vi.fn(), fillRect: vi.fn(),
+ globalAlpha: 1, fillStyle: '',
}
beforeEach(() => {
@@ -759,7 +760,9 @@ describe('VoiceInputSettings transcription test', () => {
expect(start).not.toHaveBeenCalled()
})
- it('records with the chosen device, draws the wave, and shows text, duration and timing', async () => {
+ it('records with the chosen device, draws the trace, and shows text, duration and timing', async () => {
+ // Only the clock's interval and Date: the trace's frames are driven by hand.
+ vi.useFakeTimers({ toFake: ['setInterval', 'clearInterval', 'Date'] })
localStorage.setItem('cc-haha-voice-input-device', 'mic-b')
listInputs.mockResolvedValue([
{ deviceId: 'mic-a', label: 'Built-in Microphone' },
@@ -777,21 +780,31 @@ describe('VoiceInputSettings transcription test', () => {
expect(start).toHaveBeenCalledTimes(1)
expect(start.mock.calls[0]![0]).toMatchObject({ deviceId: 'mic-b', maxSeconds: 30 })
- // The wave is decorative (canvas, aria-hidden); the wrapper carries the name for screen readers.
- const wave = screen.getByRole('img', { name: 'Microphone level' })
- expect(wave.querySelector('canvas')).toHaveAttribute('aria-hidden', 'true')
+ // The trace is decorative (canvas, aria-hidden); the wrapper carries the name for screen readers.
+ const trace = screen.getByRole('img', { name: 'Microphone level' })
+ expect(trace.querySelector('canvas')).toHaveAttribute('aria-hidden', 'true')
runFrame()
expect(recording.getLevel).toHaveBeenCalled()
- expect(canvasContext.stroke).toHaveBeenCalled()
- expect(screen.getByTestId('voice-clock')).toHaveTextContent('0:00')
+ expect(canvasContext.fill).toHaveBeenCalled()
+ expect(screen.getByTestId('voice-input-timer')).toHaveTextContent('0:00')
- now += 3_200
- runFrame()
- expect(screen.getByTestId('voice-clock')).toHaveTextContent('0:03')
+ act(() => { vi.advanceTimersByTime(3_200) })
+ expect(screen.getByTestId('voice-input-timer')).toHaveTextContent('0:03')
+ // Recording is not a failure: the composer's recording bar and this test
+ // share one look, and neither is painted in the error color.
+ expect(stopButton.closest('div')!.outerHTML).not.toMatch(/--color-error/)
- api.transcribe.mockResolvedValue(TRANSCRIPT)
+ let finishTranscription!: (value: typeof TRANSCRIPT) => void
+ api.transcribe.mockImplementation(() => new Promise((resolve) => { finishTranscription = resolve }))
fireEvent.click(stopButton)
+ // Recognition keeps the recorded trace on screen and puts its state where the clock was.
+ const recognizing = await screen.findByRole('button', { name: 'Recognizing…' })
+ expect(recognizing).toHaveAttribute('aria-busy', 'true')
+ expect(screen.getByRole('img', { name: 'Microphone level' })).toBe(trace)
+ expect(screen.queryByTestId('voice-input-timer')).toBeNull()
+ await act(async () => { finishTranscription(TRANSCRIPT) })
+
expect(await screen.findByTestId('voice-transcript')).toHaveTextContent('hello from the microphone')
expect(screen.getByText('Audio 5.6 s · Inference 0.10 s')).toBeInTheDocument()
expect(recording.stop).toHaveBeenCalledTimes(1)
diff --git a/desktop/src/pages/settings/VoiceTranscriptionTest.tsx b/desktop/src/pages/settings/VoiceTranscriptionTest.tsx
index 700bc9e0..9463f934 100644
--- a/desktop/src/pages/settings/VoiceTranscriptionTest.tsx
+++ b/desktop/src/pages/settings/VoiceTranscriptionTest.tsx
@@ -3,9 +3,10 @@ import { Mic, Square } from 'lucide-react'
import { ApiError } from '@/api/client'
import { voiceApi, type VoiceLanguage, type VoiceTranscript } from '@/api/voice'
import { Button } from '@/components/ui/Button'
+import { IconButton } from '@/components/ui/IconButton'
import { useTranslation } from '@/i18n'
import type { TranslationKey } from '@/i18n/locales/en'
-import { VoiceWave } from '@/features/voiceInput/VoiceWave'
+import { VoiceRecordingTrace } from '@/features/voiceInput/VoiceRecordingBar'
import { startRecording, type ActiveRecording, type RecordingResult } from '@/features/voiceInput/recorder'
import { recorderErrorKey, voiceErrorKey } from './useMicrophoneSelection'
@@ -15,12 +16,6 @@ function transcribeErrorKey(error: unknown): TranslationKey {
return voiceErrorKey(error instanceof ApiError ? (error.body as { error?: unknown } | null)?.error : undefined)
}
-function formatClock(totalSeconds: number): string {
- const minutes = Math.floor(totalSeconds / 60)
- const seconds = totalSeconds % 60
- return `${minutes}:${String(seconds).padStart(2, '0')}`
-}
-
type Props = {
deviceId?: string
providerId: string
@@ -35,9 +30,8 @@ type Props = {
/**
* Record a few seconds, run them through the real transcription route, and show
* what came back — the same path the composer uses, so a passing test means
- * dictation will work. The wave and clock are drawn straight from requestAnimationFrame;
- * routing 60 updates a second through React state would re-render the whole
- * result card for a purely visual element.
+ * dictation will work. It also looks like the composer's recording bar (the
+ * same trace, clock and stop disc), so the test reads as the feature it tests.
*/
export function VoiceTranscriptionTest({ deviceId, providerId, language, maxSeconds, ready, captureSupported }: Props) {
const t = useTranslation()
@@ -45,13 +39,13 @@ export function VoiceTranscriptionTest({ deviceId, providerId, language, maxSeco
const [errorKey, setErrorKey] = useState(null)
const [transcript, setTranscript] = useState(null)
const [playbackUrl, setPlaybackUrl] = useState(null)
+ const [startedAt, setStartedAt] = useState(0)
const recordingRef = useRef(null)
const startAbortRef = useRef(null)
const transcribeAbortRef = useRef(null)
const playbackUrlRef = useRef(null)
const mountedRef = useRef(true)
- const clockRef = useRef(null)
// The limit/interrupt callbacks outlive the render that created them by up to
// `maxSeconds`, so they read the latest choices from here, not from closure.
const latestRef = useRef({ providerId, language })
@@ -77,23 +71,6 @@ export function VoiceTranscriptionTest({ deviceId, providerId, language, maxSeco
}
}, [])
- useEffect(() => {
- if (phase !== 'recording') return
- const startedAt = performance.now()
- let frame = 0
- let shownSecond = -1
- const tick = () => {
- const elapsed = Math.floor((performance.now() - startedAt) / 1000)
- if (elapsed !== shownSecond && clockRef.current) {
- shownSecond = elapsed
- clockRef.current.textContent = formatClock(elapsed)
- }
- frame = requestAnimationFrame(tick)
- }
- frame = requestAnimationFrame(tick)
- return () => cancelAnimationFrame(frame)
- }, [phase])
-
const readLevel = useCallback(() => recordingRef.current?.getLevel() ?? 0, [])
const finish = async () => {
@@ -156,6 +133,7 @@ export function VoiceTranscriptionTest({ deviceId, providerId, language, maxSeco
return
}
recordingRef.current = recording
+ setStartedAt(Date.now())
setPhase('recording')
} catch (error) {
if (!mountedRef.current) return
@@ -167,7 +145,6 @@ export function VoiceTranscriptionTest({ deviceId, providerId, language, maxSeco
}
const canStart = ready && captureSupported
- const busy = phase === 'starting' || phase === 'transcribing'
return (
+ {phase === 'recording' || phase === 'transcribing' ? (
+ // One element across both phases, so recognition freezes the trace that
+ // was just recorded instead of starting a new one.
+