From 7fef344c26f17c4d7b6474a9c9eff00cdf018863 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Tue, 4 Aug 2026 00:10:56 -0700 Subject: [PATCH] [eric] voice: worklet capture with drain handshake replaces ScriptProcessor, live transcript preview in the overlay --- .../shared/voice/VoiceDictationContext.tsx | 4 +- frontend/src/shared/voice/VoiceOverlay.tsx | 32 ++++++- frontend/src/shared/voice/getPcmWorkletUrl.ts | 50 ++++++++++ .../src/shared/voice/useVoiceDictation.ts | 92 ++++++++++++++++--- frontend/src/shared/voice/voiceContext.ts | 5 +- frontend/src/types/electron.d.ts | 5 + 6 files changed, 166 insertions(+), 22 deletions(-) create mode 100644 frontend/src/shared/voice/getPcmWorkletUrl.ts diff --git a/frontend/src/shared/voice/VoiceDictationContext.tsx b/frontend/src/shared/voice/VoiceDictationContext.tsx index 06c216e1..c0f0f5b7 100644 --- a/frontend/src/shared/voice/VoiceDictationContext.tsx +++ b/frontend/src/shared/voice/VoiceDictationContext.tsx @@ -9,7 +9,7 @@ import VoiceOverlay from './VoiceOverlay'; // out-of-sync state. Mounted once near the app root. export function VoiceDictationProvider({ children }: { children: React.ReactNode }): React.ReactElement { - const { state, lastText, error, pct, feedback, toggle, start, stop, volumeRef } = useVoiceDictation(); + const { state, lastText, error, pct, feedback, partial, toggle, start, stop, volumeRef } = useVoiceDictation(); const holdMode = useAppSelector((s) => s.settings.data.voice_hold_to_talk ?? true); const dictationShortcut = useAppSelector((s) => s.settings.data.dictation_shortcut ?? null); const dictationModel = useAppSelector((s) => s.settings.data.dictation_model ?? null); @@ -62,7 +62,7 @@ export function VoiceDictationProvider({ children }: { children: React.ReactNode }, [pressStart, pressEnd]); return ( - + {children} diff --git a/frontend/src/shared/voice/VoiceOverlay.tsx b/frontend/src/shared/voice/VoiceOverlay.tsx index c64f223a..b318ba61 100644 --- a/frontend/src/shared/voice/VoiceOverlay.tsx +++ b/frontend/src/shared/voice/VoiceOverlay.tsx @@ -180,8 +180,26 @@ function feedbackIcon(icon: string): React.ReactElement { return ; } +// Live transcript preview: committed phrases solid, the in-flight hypothesis dimmed. Tail-clamped +// so the newest words are always the visible ones (openwhispr's preview overlay behavior). +const PREVIEW_TAIL_CHARS = 220; + +const LiveTranscript: React.FC<{ committed: string; tentative: string }> = ({ committed, tentative }) => { + const total = committed.length + tentative.length; + const over = total - PREVIEW_TAIL_CHARS; + const shownCommitted = over > 0 ? `…${committed.slice(Math.min(over, committed.length))}` : committed; + return ( + + {shownCommitted} + {tentative && ( + {shownCommitted ? ' ' : ''}{tentative} + )} + + ); +}; + const VoiceOverlay: React.FC = () => { - const { state, pct, feedback, volumeRef } = useVoice(); + const { state, pct, feedback, partial, volumeRef } = useVoice(); const [showFeedback, setShowFeedback] = useState(false); useEffect(() => { @@ -201,12 +219,18 @@ const VoiceOverlay: React.FC = () => { ) : null; + const hasPartial = !!partial && !!(partial.committed || partial.tentative); let content: React.ReactElement | null; if (state === 'recording') { - // The aurora IS the listening indicator; a pill on top of it read as clutter. - content = null; + // The aurora says "listening"; the card appears only once there are live words to show. + content = hasPartial ? : null; } else if (state === 'transcribing') { - content = (<>Transcribing); + content = ( + <> + + {hasPartial ? : Transcribing} + + ); } else if (state === 'preparing') { content = (<>Downloading voice model {pct}%); } else if (feedback) { diff --git a/frontend/src/shared/voice/getPcmWorkletUrl.ts b/frontend/src/shared/voice/getPcmWorkletUrl.ts new file mode 100644 index 00000000..58ea2997 --- /dev/null +++ b/frontend/src/shared/voice/getPcmWorkletUrl.ts @@ -0,0 +1,50 @@ +// OpenWhispr's capture worklet, ported near-verbatim (MIT): 800-sample Int16 buffers (50ms at +// 16kHz) posted with transferables off the audio thread, plus a "stop" -> drain -> "flushed" +// handshake so the tail of an utterance is never lost at teardown. Replaces the deprecated +// main-thread ScriptProcessorNode. +const WORKLET_SOURCE = ` +const BUFFER_SIZE = 800; +class PCMStreamingProcessor extends AudioWorkletProcessor { + constructor() { + super(); + this.buffer = new Int16Array(BUFFER_SIZE); + this.offset = 0; + this.stopped = false; + this.port.onmessage = (event) => { + if (event.data === "stop") { + if (this.offset > 0) { + const partial = this.buffer.slice(0, this.offset); + this.port.postMessage(partial.buffer, [partial.buffer]); + this.buffer = new Int16Array(BUFFER_SIZE); + this.offset = 0; + } + this.port.postMessage("flushed"); + this.stopped = true; + } + }; + } + process(inputs) { + if (this.stopped) return false; + const input = inputs[0] && inputs[0][0]; + if (!input) return true; + for (let i = 0; i < input.length; i++) { + const s = Math.max(-1, Math.min(1, input[i])); + this.buffer[this.offset++] = s < 0 ? s * 0x8000 : s * 0x7fff; + if (this.offset >= BUFFER_SIZE) { + this.port.postMessage(this.buffer.buffer, [this.buffer.buffer]); + this.buffer = new Int16Array(BUFFER_SIZE); + this.offset = 0; + } + } + return true; + } +} +registerProcessor("pcm-streaming-processor", PCMStreamingProcessor); +`; + +let cachedUrl: string | null = null; + +export function getPcmWorkletUrl(): string { + if (!cachedUrl) cachedUrl = URL.createObjectURL(new Blob([WORKLET_SOURCE], { type: 'application/javascript' })); + return cachedUrl; +} diff --git a/frontend/src/shared/voice/useVoiceDictation.ts b/frontend/src/shared/voice/useVoiceDictation.ts index 0aa36472..e8748037 100644 --- a/frontend/src/shared/voice/useVoiceDictation.ts +++ b/frontend/src/shared/voice/useVoiceDictation.ts @@ -4,19 +4,30 @@ import { encodeWav, VOICE_SAMPLE_RATE } from './encodeWav'; import { playVoiceCue } from './voiceCues'; import { injectAtFocus } from './injectAtFocus'; import { createSilenceDetector } from './createSilenceDetector'; +import { getPcmWorkletUrl } from './getPcmWorkletUrl'; export type VoiceState = 'idle' | 'recording' | 'transcribing' | 'preparing'; +// Live transcript preview: committed phrases are decoded once and never rewritten; the tentative +// tail is the latest hypothesis for the phrase still being spoken. +export interface VoicePartial { + committed: string; + tentative: string; +} + // WhisperFlow-style push-to-dictate: toggle recording (global hotkey or a mic), speak, and the -// transcribed text is pasted into whatever field has focus. Capture is 16kHz mono PCM so it feeds -// whisper.cpp with no server-side resample. The recording path can only be proven with a real mic; +// transcribed text is pasted into whatever field has focus. Capture is 16kHz mono PCM off the audio +// thread (AudioWorklet) so it feeds whisper.cpp with no server-side resample, and the same chunks +// stream to main for live partials. The recording path can only be proven with a real mic; // the encode -> transcribe -> inject half is exercised by the encoder round-trip test. interface Recorder { ctx: AudioContext; stream: MediaStream; - node: ScriptProcessorNode; + node: AudioWorkletNode; source: MediaStreamAudioSourceNode; chunks: Float32Array[]; + flushed: Promise; + streaming: boolean; } // One object per terminal outcome so the overlay's effect always re-fires (new identity every time). @@ -60,6 +71,8 @@ export function useVoiceDictation() { const [error, setError] = useState(null); const [pct, setPct] = useState(0); const [feedback, setFeedback] = useState(null); + const [partial, setPartial] = useState(null); + const partialSeqRef = useRef(0); const recRef = useRef(null); const stateRef = useRef('idle'); // Live mic level (0..1) for the aurora; a ref, not state, so 60Hz visuals never re-render React. @@ -104,19 +117,33 @@ export function useVoiceDictation() { if (stateRef.current !== 'idle') return; if (!window.openswarm?.voiceTranscribe) { setError('desktop-only'); return; } // no Electron bridge = web build setError(null); + // Warm on the DOWN edge, before the mic prompt, so the model load overlaps the user starting to speak. + void window.openswarm?.voiceWarmup?.(); + setPartial(null); + partialSeqRef.current = 0; try { // Fire the OS mic prompt through the main process first: a packaged hardened-runtime build denies renderer getUserMedia outright until TCC granted (the prod dictation-dead cause, ENG-103). const micOk = await (window.openswarm as any)?.voiceRequestMicAccess?.() ?? true; if (micOk === false) { setError('mic-denied'); return; } const stream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } }); const ctx = new AudioContext({ sampleRate: VOICE_SAMPLE_RATE }); + await ctx.audioWorklet.addModule(getPcmWorkletUrl()); const source = ctx.createMediaStreamSource(stream); - const node = ctx.createScriptProcessor(4096, 1, 1); + const node = new AudioWorkletNode(ctx, 'pcm-streaming-processor'); const chunks: Float32Array[] = []; const endpointer = hold ? null : createSilenceDetector(ctx.sampleRate); - node.onaudioprocess = (e): void => { - const data = e.inputBuffer.getChannelData(0); - chunks.push(new Float32Array(data)); + const streamRes = await window.openswarm?.voiceStreamStart?.(); + const streaming = streamRes?.ok === true; + let flushResolve: () => void = () => {}; + const flushed = new Promise((resolve) => { flushResolve = resolve; }); + node.port.onmessage = (e: MessageEvent): void => { + if (e.data === 'flushed') { flushResolve(); return; } + const ab = e.data as ArrayBuffer; + if (streaming) window.openswarm?.voiceStreamChunk?.(ab); + const i16 = new Int16Array(ab); + const data = new Float32Array(i16.length); + for (let i = 0; i < i16.length; i++) data[i] = i16[i] / 0x8000; + chunks.push(data); // RMS per chunk drives the aurora; smoothed so it breathes instead of flickering. let sum = 0; for (let i = 0; i < data.length; i += 8) sum += data[i] * data[i]; @@ -126,12 +153,10 @@ export function useVoiceDictation() { }; source.connect(node); node.connect(ctx.destination); - recRef.current = { ctx, stream, node, source, chunks }; + recRef.current = { ctx, stream, node, source, chunks, flushed, streaming }; setState('recording'); playVoiceCue('start'); void (window.openswarm as { haptic?: (p: string) => Promise } | undefined)?.haptic?.('generic'); - // Warm the model the moment recording begins so transcription is instant on stop. - void window.openswarm?.voiceWarmup?.(); } catch (err) { const msg = err instanceof Error ? err.message : 'mic-unavailable'; setError(msg); @@ -143,14 +168,35 @@ export function useVoiceDictation() { const stop = useCallback(async (): Promise => { if (stateRef.current !== 'recording') return; + const rec = recRef.current; + const streaming = rec?.streaming === true; + // Drain the worklet's last partial buffer before teardown (1s watchdog: a wedged worklet costs 50ms of tail, not a hang). + if (rec) { + try { rec.node.port.postMessage('stop'); } catch (_) { /* node already gone */ } + await Promise.race([rec.flushed, new Promise((resolve) => window.setTimeout(resolve, 1000))]); + } const samples = teardown(); playVoiceCue('stop'); void (window.openswarm as { haptic?: (p: string) => Promise } | undefined)?.haptic?.('alignment'); setState('transcribing'); try { - if (!samples || samples.length < VOICE_SAMPLE_RATE * 0.2) { setState('idle'); return; } // < 0.2s = a misfire - const wav = encodeWav(samples); - const res = await window.openswarm?.voiceTranscribe?.(wav); + if (!samples || samples.length < VOICE_SAMPLE_RATE * 0.2) { // < 0.2s = a misfire + if (streaming) window.openswarm?.voiceStreamCancel?.(); + setPartial(null); + setState('idle'); + return; + } + // The streamed assembly (each phrase decoded once at its boundary) is the fast path: stop only + // pays for the final open phrase. Any doubt (degraded, empty) falls back to one full-clip decode. + let res: { ok: boolean; text?: string; error?: string } | undefined; + if (streaming) { + const sres = await window.openswarm?.voiceStreamStop?.(); + if (sres?.ok && sres.text && !sres.degraded) res = { ok: true, text: sres.text }; + } + if (!res) { + const wav = encodeWav(samples); + res = await window.openswarm?.voiceTranscribe?.(wav); + } if (res?.ok && res.text) { // WhisperFlow-style cleanup: punctuation + filler words via the cheap aux tier, fail-open to // the raw transcript on any error/timeout so dictation never breaks with the aux down. @@ -203,8 +249,26 @@ export function useVoiceDictation() { return () => { off?.(); }; }, [toggle]); + // Live partials from the main-process streaming session; the seq guard drops anything a stale + // session (previous recording's in-flight decode) manages to emit after a new one started. + useEffect(() => { + const off = window.openswarm?.onVoicePartial?.((p) => { + if (p.seq <= partialSeqRef.current) return; + partialSeqRef.current = p.seq; + if (stateRef.current === 'recording' || stateRef.current === 'transcribing') { + setPartial({ committed: p.committed, tentative: p.tentative }); + } + }); + return () => { off?.(); }; + }, []); + + // The preview belongs to a live session only; any terminal state clears it in one place. + useEffect(() => { + if (state === 'idle' || state === 'preparing') setPartial(null); + }, [state]); + // A dangling recorder (unmount mid-capture) must release the mic. useEffect(() => () => { teardown(); }, [teardown]); - return { state, lastText, error, pct, feedback, toggle, start, stop, volumeRef }; + return { state, lastText, error, pct, feedback, partial, toggle, start, stop, volumeRef }; } diff --git a/frontend/src/shared/voice/voiceContext.ts b/frontend/src/shared/voice/voiceContext.ts index 47cc6458..582382c0 100644 --- a/frontend/src/shared/voice/voiceContext.ts +++ b/frontend/src/shared/voice/voiceContext.ts @@ -1,5 +1,5 @@ import React, { createContext, useContext } from 'react'; -import { VoiceState, VoiceFeedback } from './useVoiceDictation'; +import { VoiceState, VoiceFeedback, VoicePartial } from './useVoiceDictation'; // The context lives below both the provider and the overlay so neither imports the other // (VoiceDictationContext renders VoiceOverlay; both reach down here instead of sideways). @@ -9,6 +9,7 @@ export interface VoiceContextValue { error: string | null; pct: number; feedback: VoiceFeedback | null; + partial: VoicePartial | null; toggle: () => void; // Mic-button press semantics that respect the hold/toggle setting: press starts (or toggles), // release stops only in hold mode. Buttons wire onPointerDown/Up to these and stay mode-agnostic. @@ -19,7 +20,7 @@ export interface VoiceContextValue { } const NOOP_REF = { current: 0 }; -const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, feedback: null, toggle: () => {}, pressStart: () => {}, pressEnd: () => {}, holdMode: true, volumeRef: NOOP_REF }; +const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, feedback: null, partial: null, toggle: () => {}, pressStart: () => {}, pressEnd: () => {}, holdMode: true, volumeRef: NOOP_REF }; export const VoiceContext = createContext(NOOP); diff --git a/frontend/src/types/electron.d.ts b/frontend/src/types/electron.d.ts index 9d782dde..9268e15b 100644 --- a/frontend/src/types/electron.d.ts +++ b/frontend/src/types/electron.d.ts @@ -86,6 +86,11 @@ declare global { voiceSetModel?: (id: string) => Promise<{ ok: boolean; ready: boolean }>; voiceTranscribe?: (wav: ArrayBuffer) => Promise<{ ok: boolean; text?: string; error?: string }>; voiceInject?: (text: string) => Promise<{ ok: boolean; pasted?: boolean; error?: string }>; + voiceStreamStart?: () => Promise<{ ok: boolean; error?: string }>; + voiceStreamChunk?: (pcm: ArrayBuffer) => void; + voiceStreamStop?: () => Promise<{ ok: boolean; text?: string; degraded?: boolean; error?: string }>; + voiceStreamCancel?: () => void; + onVoicePartial?: (cb: (p: { committed: string; tentative: string; seq: number }) => void) => () => void; onVoiceToggle?: (cb: () => void) => () => void; voiceHoldCapable?: () => Promise; voiceRequestHoldPermission?: () => Promise;