[eric] voice: fix stuck model-downloading state + harden download + visible status overlay

This commit is contained in:
ciregenz
2026-07-21 23:59:47 -07:00
parent 8dfe6fbda4
commit 17364d5439
4 changed files with 194 additions and 55 deletions
@@ -1,5 +1,6 @@
import React, { createContext, useContext } from 'react';
import { useVoiceDictation, VoiceState } from './useVoiceDictation';
import { useVoiceDictation, VoiceState, VoiceFeedback } from './useVoiceDictation';
import VoiceOverlay from './VoiceOverlay';
// One recorder for the whole app. Both mics (the Help pill and the spawn composer) plus the global
// hotkey drive the SAME dictation session, so two mics can't fight over the microphone or show
@@ -9,17 +10,19 @@ interface VoiceContextValue {
lastText: string;
error: string | null;
pct: number;
feedback: VoiceFeedback | null;
toggle: () => void;
}
const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, toggle: () => {} };
const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, feedback: null, toggle: () => {} };
const VoiceContext = createContext<VoiceContextValue>(NOOP);
export function VoiceDictationProvider({ children }: { children: React.ReactNode }): React.ReactElement {
const { state, lastText, error, pct, toggle } = useVoiceDictation();
const { state, lastText, error, pct, feedback, toggle } = useVoiceDictation();
return (
<VoiceContext.Provider value={{ state, lastText, error, pct, toggle }}>
<VoiceContext.Provider value={{ state, lastText, error, pct, feedback, toggle }}>
{children}
<VoiceOverlay />
</VoiceContext.Provider>
);
}
@@ -0,0 +1,99 @@
import React, { useEffect, useState } from 'react';
import Box from '@mui/material/Box';
import CircularProgress from '@mui/material/CircularProgress';
import MicIcon from '@mui/icons-material/Mic';
import CheckRoundedIcon from '@mui/icons-material/CheckRounded';
import ContentPasteRoundedIcon from '@mui/icons-material/ContentPasteRounded';
import InfoOutlinedIcon from '@mui/icons-material/InfoOutlined';
import { useVoice } from './VoiceDictationContext';
// The whole point: dictation must never look like "nothing happened." This floats a small status
// card above the composer for every phase (listening, transcribing, downloading the model) and shows
// the transcript + whether it was pasted or just copied. Non-interactive, auto-dismisses.
const FEEDBACK_MS = 4500;
function feedbackIcon(icon: string): React.ReactElement {
if (icon === 'check') return <CheckRoundedIcon sx={{ fontSize: 16, color: '#4ade80' }} />;
if (icon === 'clipboard') return <ContentPasteRoundedIcon sx={{ fontSize: 15, color: 'rgba(255,255,255,0.8)' }} />;
if (icon === 'mic') return <MicIcon sx={{ fontSize: 16, color: '#ff8a8a' }} />;
return <InfoOutlinedIcon sx={{ fontSize: 15, color: 'rgba(255,255,255,0.8)' }} />;
}
const VoiceOverlay: React.FC = () => {
const { state, pct, feedback } = useVoice();
const [showFeedback, setShowFeedback] = useState(false);
useEffect(() => {
if (!feedback) return undefined;
setShowFeedback(true);
const t = setTimeout(() => setShowFeedback(false), FEEDBACK_MS);
return () => clearTimeout(t);
}, [feedback]);
const live = state !== 'idle';
const visible = live || (showFeedback && !!feedback);
if (!visible) return null;
let content: React.ReactElement;
if (state === 'recording') {
content = (
<>
<MicIcon sx={{ fontSize: 16, color: '#ff8a8a' }} />
<span>Listening</span>
<Box component="span" sx={{
width: 6, height: 6, borderRadius: '50%', background: '#ff8a8a', ml: 0.25,
'@keyframes vpulse': { '0%,100%': { opacity: 0.3 }, '50%': { opacity: 1 } },
animation: 'vpulse 1s ease-in-out infinite',
}} />
</>
);
} else if (state === 'transcribing') {
content = (<><CircularProgress size={13} thickness={5} sx={{ color: 'rgba(255,255,255,0.7)' }} /><span>Transcribing</span></>);
} else if (state === 'preparing') {
content = (<><CircularProgress size={13} thickness={5} sx={{ color: 'rgba(255,255,255,0.7)' }} /><span>Downloading voice model {pct}%</span></>);
} else if (feedback) {
content = (
<>
{feedbackIcon(feedback.icon)}
<Box component="span" sx={{ maxWidth: 420, overflow: 'hidden', textOverflow: 'ellipsis', whiteSpace: 'nowrap' }}>
{feedback.text}
</Box>
</>
);
} else {
return null;
}
return (
<Box
sx={{
position: 'fixed',
bottom: 84,
left: '50%',
transform: 'translateX(-50%)',
zIndex: 2147483000,
pointerEvents: 'none',
display: 'flex',
alignItems: 'center',
gap: 1,
px: 1.75,
py: 0.9,
maxWidth: '80vw',
borderRadius: 999,
background: 'rgba(22,12,34,0.9)',
backdropFilter: 'blur(20px) saturate(160%)',
WebkitBackdropFilter: 'blur(20px) saturate(160%)',
boxShadow: '0 8px 28px rgba(0,0,0,0.4)',
color: 'rgba(255,255,255,0.92)',
fontSize: '0.82rem',
fontWeight: 500,
'@keyframes vin': { from: { opacity: 0, transform: 'translate(-50%, 6px)' }, to: { opacity: 1, transform: 'translate(-50%, 0)' } },
animation: 'vin 0.16s ease-out',
}}
>
{content}
</Box>
);
};
export default VoiceOverlay;
+23 -3
View File
@@ -15,11 +15,20 @@ interface Recorder {
chunks: Float32Array[];
}
// One object per terminal outcome so the overlay's effect always re-fires (new identity every time).
export interface VoiceFeedback {
tone: 'ok' | 'warn' | 'error';
icon: 'check' | 'clipboard' | 'mic' | 'info';
text: string;
at: number;
}
export function useVoiceDictation() {
const [state, setState] = useState<VoiceState>('idle');
const [lastText, setLastText] = useState<string>('');
const [error, setError] = useState<string | null>(null);
const [pct, setPct] = useState<number>(0);
const [feedback, setFeedback] = useState<VoiceFeedback | null>(null);
const recRef = useRef<Recorder | null>(null);
const stateRef = useRef<VoiceState>('idle');
stateRef.current = state;
@@ -72,7 +81,10 @@ export function useVoiceDictation() {
// Warm the model the moment recording begins so transcription is instant on stop.
void window.openswarm?.voiceWarmup?.();
} catch (err) {
setError(err instanceof Error ? err.message : 'mic-unavailable');
const msg = err instanceof Error ? err.message : 'mic-unavailable';
setError(msg);
const denied = /NotAllowed|Permission|denied/i.test(msg);
setFeedback({ tone: 'error', icon: 'mic', text: denied ? 'Microphone access needed. Enable it in System Settings, Privacy, Microphone.' : 'Could not start the microphone.', at: Date.now() });
setState('idle');
}
}, []);
@@ -87,7 +99,13 @@ export function useVoiceDictation() {
const res = await window.openswarm?.voiceTranscribe?.(wav);
if (res?.ok && res.text) {
setLastText(res.text);
await window.openswarm?.voiceInject?.(res.text);
const inj = await window.openswarm?.voiceInject?.(res.text);
setFeedback(inj?.pasted
? { tone: 'ok', icon: 'check', text: res.text, at: Date.now() }
: { tone: 'ok', icon: 'clipboard', text: `${res.text} (copied, press Cmd+V)`, at: Date.now() });
setState('idle');
} else if (res?.ok && !res.text) {
setFeedback({ tone: 'warn', icon: 'info', text: "Didn't catch that. Try again.", at: Date.now() });
setState('idle');
} else if (res?.error === 'model-downloading' || res?.error === 'no-model') {
// First use kicked off the model fetch; show progress and don't error out.
@@ -95,10 +113,12 @@ export function useVoiceDictation() {
pollModel();
} else {
setError(res?.error || 'transcription-failed');
setFeedback({ tone: 'error', icon: 'info', text: 'Voice transcription failed. Try again.', at: Date.now() });
setState('idle');
}
} catch (err) {
setError(err instanceof Error ? err.message : 'transcription-failed');
setFeedback({ tone: 'error', icon: 'info', text: 'Voice transcription failed. Try again.', at: Date.now() });
setState('idle');
}
}, [teardown, pollModel]);
@@ -117,5 +137,5 @@ export function useVoiceDictation() {
// A dangling recorder (unmount mid-capture) must release the mic.
useEffect(() => () => { teardown(); }, [teardown]);
return { state, lastText, error, pct, toggle, start, stop };
return { state, lastText, error, pct, feedback, toggle, start, stop };
}