diff --git a/frontend/src/app/pages/Dashboard/desktop/DesktopSpawnPill.tsx b/frontend/src/app/pages/Dashboard/desktop/DesktopSpawnPill.tsx index 264cf8b6..59c78b72 100644 --- a/frontend/src/app/pages/Dashboard/desktop/DesktopSpawnPill.tsx +++ b/frontend/src/app/pages/Dashboard/desktop/DesktopSpawnPill.tsx @@ -40,7 +40,7 @@ function DesktopSpawnPill({ const [menuOpen, setMenuOpen] = useState(false); const newAgentShortcut = useAppSelector((s) => s.settings.data.new_agent_shortcut); const rootRef = useRef(null); - const { state: voiceState, pct: voicePct, pressStart: voicePressStart, pressEnd: voicePressEnd } = useVoice(); + const { state: voiceState, pct: voicePct, pressStart: voicePressStart, pressEnd: voicePressEnd, prewarm: voicePrewarm } = useVoice(); const recording = voiceState === 'recording'; const transcribing = voiceState === 'transcribing'; const preparing = voiceState === 'preparing'; @@ -166,7 +166,10 @@ function DesktopSpawnPill({ { e.stopPropagation(); if (!voiceBusy) voicePressStart(); }} + // Open the mic on hover so the FIRST press of a launch is warm. Cold it costs ~2.5s and + // silently eats what you say, which reads as "the button did nothing" (ENG-300). + onPointerEnter={() => { if (!voiceBusy) voicePrewarm(); }} + onPointerDown={(e) => { e.stopPropagation(); if (!voiceBusy) voicePressStart(); }} onPointerUp={(e) => { e.stopPropagation(); voicePressEnd(); }} onPointerLeave={() => voicePressEnd()} onClick={(e) => e.stopPropagation()} diff --git a/frontend/src/shared/voice/VoiceDictationContext.tsx b/frontend/src/shared/voice/VoiceDictationContext.tsx index 960e7384..6ceed60d 100644 --- a/frontend/src/shared/voice/VoiceDictationContext.tsx +++ b/frontend/src/shared/voice/VoiceDictationContext.tsx @@ -12,7 +12,7 @@ import VoiceOverlay from './VoiceOverlay'; // out-of-sync state. Mounted once near the app root. export function VoiceDictationProvider({ children }: { children: React.ReactNode }): React.ReactElement { - const { state, lastText, error, pct, feedback, partial, target, toggle, start, stop, cancel, notify, volumeRef } = useVoiceDictation(); + const { state, lastText, error, pct, feedback, partial, target, toggle, start, stop, cancel, notify, prewarm, volumeRef } = useVoiceDictation(); // fn is the dictation key, but macOS may still have its own Globe action bound (emoji picker on a quick tap); say so once. const globeWarnedRef = useRef(false); @@ -136,7 +136,7 @@ export function VoiceDictationProvider({ children }: { children: React.ReactNode const confirmRecording = useCallback((): void => { void stop(); }, [stop]); return ( - + { void prewarm(); }, pressStart, pressEnd, confirmRecording, cancelRecording: cancel, holdMode, volumeRef }}> {children} diff --git a/frontend/src/shared/voice/useVoiceDictation.ts b/frontend/src/shared/voice/useVoiceDictation.ts index 067d419a..f01e0c98 100644 --- a/frontend/src/shared/voice/useVoiceDictation.ts +++ b/frontend/src/shared/voice/useVoiceDictation.ts @@ -169,6 +169,25 @@ export function useVoiceDictation() { useEffect(() => () => releaseWarmMic(), [releaseWarmMic]); + // ENG-300: the park above only helps AFTER a first session, so the FIRST press of every launch + // paid the full ~2.5s cold open and dropped whatever the user said into it. Measured by Eric: the + // button "never turns on until I unclick it", every time, right after startup. Arming on hover + // makes that first press warm like the rest, and costs nothing until the user reaches for the + // control, which is why this is not done at boot: an always-hot mic lights the OS indicator for + // people who never dictate. + const prewarm = useCallback(async (): Promise => { + if (warmMicRef.current || stateRef.current !== 'idle') return; + try { + const stream = await navigator.mediaDevices.getUserMedia({ + audio: { channelCount: 1, echoCancellation: true, noiseSuppression: false, autoGainControl: false }, + }); + parkWarmMic(stream, new AudioContext({ sampleRate: VOICE_SAMPLE_RATE })); + } catch (_) { + // No permission yet, or no device. Stay silent: the real press is what should surface that, + // and a hover must never raise an error the user did not ask for. + } + }, [parkWarmMic]); + const teardown = useCallback((): Float32Array | null => { const rec = recRef.current; recRef.current = null; @@ -460,5 +479,5 @@ export function useVoiceDictation() { setFeedback({ tone: 'warn', icon: 'info', text, at: Date.now() }); }, []); - return { state, lastText, error, pct, feedback, partial, target, toggle, start, stop, cancel, notify, volumeRef }; + return { state, lastText, error, pct, feedback, partial, target, toggle, start, stop, cancel, notify, prewarm, volumeRef }; } diff --git a/frontend/src/shared/voice/voiceContext.ts b/frontend/src/shared/voice/voiceContext.ts index 559d8a3f..8552e77b 100644 --- a/frontend/src/shared/voice/voiceContext.ts +++ b/frontend/src/shared/voice/voiceContext.ts @@ -13,6 +13,9 @@ export interface VoiceContextValue { // Where the transcript will land right now, in user words plus the surface's icon. target: InjectTargetInfo; toggle: () => void; + // Open and park the mic before the first press, so it is not cold when the user actually + // clicks (ENG-300). Safe to call repeatedly; a no-op once armed or while recording. + prewarm: () => void; // Mic-button press semantics that respect the hold/toggle setting: press starts (or toggles), // release stops only in hold mode. Buttons wire onPointerDown/Up to these and stay mode-agnostic. pressStart: () => void; @@ -25,7 +28,7 @@ export interface VoiceContextValue { } const NOOP_REF = { current: 0 }; -const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, feedback: null, partial: null, target: { label: '', icon: null, composerId: null }, toggle: () => {}, pressStart: () => {}, pressEnd: () => {}, confirmRecording: () => {}, cancelRecording: () => {}, holdMode: true, volumeRef: NOOP_REF }; +const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, feedback: null, partial: null, target: { label: '', icon: null, composerId: null }, toggle: () => {}, prewarm: () => {}, pressStart: () => {}, pressEnd: () => {}, confirmRecording: () => {}, cancelRecording: () => {}, holdMode: true, volumeRef: NOOP_REF }; export const VoiceContext = createContext(NOOP);