mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-10-01 05:54:56 +02:00
[eric] voice: fix stuck model-downloading state + harden download + visible status overlay
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
import React, { createContext, useContext } from 'react';
|
||||
import { useVoiceDictation, VoiceState } from './useVoiceDictation';
|
||||
import { useVoiceDictation, VoiceState, VoiceFeedback } from './useVoiceDictation';
|
||||
import VoiceOverlay from './VoiceOverlay';
|
||||
|
||||
// One recorder for the whole app. Both mics (the Help pill and the spawn composer) plus the global
|
||||
// hotkey drive the SAME dictation session, so two mics can't fight over the microphone or show
|
||||
@@ -9,17 +10,19 @@ interface VoiceContextValue {
|
||||
lastText: string;
|
||||
error: string | null;
|
||||
pct: number;
|
||||
feedback: VoiceFeedback | null;
|
||||
toggle: () => void;
|
||||
}
|
||||
|
||||
const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, toggle: () => {} };
|
||||
const NOOP: VoiceContextValue = { state: 'idle', lastText: '', error: null, pct: 0, feedback: null, toggle: () => {} };
|
||||
const VoiceContext = createContext<VoiceContextValue>(NOOP);
|
||||
|
||||
export function VoiceDictationProvider({ children }: { children: React.ReactNode }): React.ReactElement {
|
||||
const { state, lastText, error, pct, toggle } = useVoiceDictation();
|
||||
const { state, lastText, error, pct, feedback, toggle } = useVoiceDictation();
|
||||
return (
|
||||
<VoiceContext.Provider value={{ state, lastText, error, pct, toggle }}>
|
||||
<VoiceContext.Provider value={{ state, lastText, error, pct, feedback, toggle }}>
|
||||
{children}
|
||||
<VoiceOverlay />
|
||||
</VoiceContext.Provider>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
import React, { useEffect, useState } from 'react';
|
||||
import Box from '@mui/material/Box';
|
||||
import CircularProgress from '@mui/material/CircularProgress';
|
||||
import MicIcon from '@mui/icons-material/Mic';
|
||||
import CheckRoundedIcon from '@mui/icons-material/CheckRounded';
|
||||
import ContentPasteRoundedIcon from '@mui/icons-material/ContentPasteRounded';
|
||||
import InfoOutlinedIcon from '@mui/icons-material/InfoOutlined';
|
||||
import { useVoice } from './VoiceDictationContext';
|
||||
|
||||
// The whole point: dictation must never look like "nothing happened." This floats a small status
|
||||
// card above the composer for every phase (listening, transcribing, downloading the model) and shows
|
||||
// the transcript + whether it was pasted or just copied. Non-interactive, auto-dismisses.
|
||||
const FEEDBACK_MS = 4500;
|
||||
|
||||
function feedbackIcon(icon: string): React.ReactElement {
|
||||
if (icon === 'check') return <CheckRoundedIcon sx={{ fontSize: 16, color: '#4ade80' }} />;
|
||||
if (icon === 'clipboard') return <ContentPasteRoundedIcon sx={{ fontSize: 15, color: 'rgba(255,255,255,0.8)' }} />;
|
||||
if (icon === 'mic') return <MicIcon sx={{ fontSize: 16, color: '#ff8a8a' }} />;
|
||||
return <InfoOutlinedIcon sx={{ fontSize: 15, color: 'rgba(255,255,255,0.8)' }} />;
|
||||
}
|
||||
|
||||
const VoiceOverlay: React.FC = () => {
|
||||
const { state, pct, feedback } = useVoice();
|
||||
const [showFeedback, setShowFeedback] = useState(false);
|
||||
|
||||
useEffect(() => {
|
||||
if (!feedback) return undefined;
|
||||
setShowFeedback(true);
|
||||
const t = setTimeout(() => setShowFeedback(false), FEEDBACK_MS);
|
||||
return () => clearTimeout(t);
|
||||
}, [feedback]);
|
||||
|
||||
const live = state !== 'idle';
|
||||
const visible = live || (showFeedback && !!feedback);
|
||||
if (!visible) return null;
|
||||
|
||||
let content: React.ReactElement;
|
||||
if (state === 'recording') {
|
||||
content = (
|
||||
<>
|
||||
<MicIcon sx={{ fontSize: 16, color: '#ff8a8a' }} />
|
||||
<span>Listening</span>
|
||||
<Box component="span" sx={{
|
||||
width: 6, height: 6, borderRadius: '50%', background: '#ff8a8a', ml: 0.25,
|
||||
'@keyframes vpulse': { '0%,100%': { opacity: 0.3 }, '50%': { opacity: 1 } },
|
||||
animation: 'vpulse 1s ease-in-out infinite',
|
||||
}} />
|
||||
</>
|
||||
);
|
||||
} else if (state === 'transcribing') {
|
||||
content = (<><CircularProgress size={13} thickness={5} sx={{ color: 'rgba(255,255,255,0.7)' }} /><span>Transcribing</span></>);
|
||||
} else if (state === 'preparing') {
|
||||
content = (<><CircularProgress size={13} thickness={5} sx={{ color: 'rgba(255,255,255,0.7)' }} /><span>Downloading voice model {pct}%</span></>);
|
||||
} else if (feedback) {
|
||||
content = (
|
||||
<>
|
||||
{feedbackIcon(feedback.icon)}
|
||||
<Box component="span" sx={{ maxWidth: 420, overflow: 'hidden', textOverflow: 'ellipsis', whiteSpace: 'nowrap' }}>
|
||||
{feedback.text}
|
||||
</Box>
|
||||
</>
|
||||
);
|
||||
} else {
|
||||
return null;
|
||||
}
|
||||
|
||||
return (
|
||||
<Box
|
||||
sx={{
|
||||
position: 'fixed',
|
||||
bottom: 84,
|
||||
left: '50%',
|
||||
transform: 'translateX(-50%)',
|
||||
zIndex: 2147483000,
|
||||
pointerEvents: 'none',
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
gap: 1,
|
||||
px: 1.75,
|
||||
py: 0.9,
|
||||
maxWidth: '80vw',
|
||||
borderRadius: 999,
|
||||
background: 'rgba(22,12,34,0.9)',
|
||||
backdropFilter: 'blur(20px) saturate(160%)',
|
||||
WebkitBackdropFilter: 'blur(20px) saturate(160%)',
|
||||
boxShadow: '0 8px 28px rgba(0,0,0,0.4)',
|
||||
color: 'rgba(255,255,255,0.92)',
|
||||
fontSize: '0.82rem',
|
||||
fontWeight: 500,
|
||||
'@keyframes vin': { from: { opacity: 0, transform: 'translate(-50%, 6px)' }, to: { opacity: 1, transform: 'translate(-50%, 0)' } },
|
||||
animation: 'vin 0.16s ease-out',
|
||||
}}
|
||||
>
|
||||
{content}
|
||||
</Box>
|
||||
);
|
||||
};
|
||||
|
||||
export default VoiceOverlay;
|
||||
@@ -15,11 +15,20 @@ interface Recorder {
|
||||
chunks: Float32Array[];
|
||||
}
|
||||
|
||||
// One object per terminal outcome so the overlay's effect always re-fires (new identity every time).
|
||||
export interface VoiceFeedback {
|
||||
tone: 'ok' | 'warn' | 'error';
|
||||
icon: 'check' | 'clipboard' | 'mic' | 'info';
|
||||
text: string;
|
||||
at: number;
|
||||
}
|
||||
|
||||
export function useVoiceDictation() {
|
||||
const [state, setState] = useState<VoiceState>('idle');
|
||||
const [lastText, setLastText] = useState<string>('');
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [pct, setPct] = useState<number>(0);
|
||||
const [feedback, setFeedback] = useState<VoiceFeedback | null>(null);
|
||||
const recRef = useRef<Recorder | null>(null);
|
||||
const stateRef = useRef<VoiceState>('idle');
|
||||
stateRef.current = state;
|
||||
@@ -72,7 +81,10 @@ export function useVoiceDictation() {
|
||||
// Warm the model the moment recording begins so transcription is instant on stop.
|
||||
void window.openswarm?.voiceWarmup?.();
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : 'mic-unavailable');
|
||||
const msg = err instanceof Error ? err.message : 'mic-unavailable';
|
||||
setError(msg);
|
||||
const denied = /NotAllowed|Permission|denied/i.test(msg);
|
||||
setFeedback({ tone: 'error', icon: 'mic', text: denied ? 'Microphone access needed. Enable it in System Settings, Privacy, Microphone.' : 'Could not start the microphone.', at: Date.now() });
|
||||
setState('idle');
|
||||
}
|
||||
}, []);
|
||||
@@ -87,7 +99,13 @@ export function useVoiceDictation() {
|
||||
const res = await window.openswarm?.voiceTranscribe?.(wav);
|
||||
if (res?.ok && res.text) {
|
||||
setLastText(res.text);
|
||||
await window.openswarm?.voiceInject?.(res.text);
|
||||
const inj = await window.openswarm?.voiceInject?.(res.text);
|
||||
setFeedback(inj?.pasted
|
||||
? { tone: 'ok', icon: 'check', text: res.text, at: Date.now() }
|
||||
: { tone: 'ok', icon: 'clipboard', text: `${res.text} (copied, press Cmd+V)`, at: Date.now() });
|
||||
setState('idle');
|
||||
} else if (res?.ok && !res.text) {
|
||||
setFeedback({ tone: 'warn', icon: 'info', text: "Didn't catch that. Try again.", at: Date.now() });
|
||||
setState('idle');
|
||||
} else if (res?.error === 'model-downloading' || res?.error === 'no-model') {
|
||||
// First use kicked off the model fetch; show progress and don't error out.
|
||||
@@ -95,10 +113,12 @@ export function useVoiceDictation() {
|
||||
pollModel();
|
||||
} else {
|
||||
setError(res?.error || 'transcription-failed');
|
||||
setFeedback({ tone: 'error', icon: 'info', text: 'Voice transcription failed. Try again.', at: Date.now() });
|
||||
setState('idle');
|
||||
}
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : 'transcription-failed');
|
||||
setFeedback({ tone: 'error', icon: 'info', text: 'Voice transcription failed. Try again.', at: Date.now() });
|
||||
setState('idle');
|
||||
}
|
||||
}, [teardown, pollModel]);
|
||||
@@ -117,5 +137,5 @@ export function useVoiceDictation() {
|
||||
// A dangling recorder (unmount mid-capture) must release the mic.
|
||||
useEffect(() => () => { teardown(); }, [teardown]);
|
||||
|
||||
return { state, lastText, error, pct, toggle, start, stop };
|
||||
return { state, lastText, error, pct, feedback, toggle, start, stop };
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user