diff --git a/backend/apps/settings/models.py b/backend/apps/settings/models.py index 384677c5..8c691fe9 100644 --- a/backend/apps/settings/models.py +++ b/backend/apps/settings/models.py @@ -52,6 +52,8 @@ class AppSettings(BaseModel): # None = platform default (Cmd/Ctrl+Shift+D); parts format matches new_agent_shortcut. dictation_shortcut: Optional[str] = None voice_hold_to_talk: bool = True + # Whisper model id from the desktop catalog (electron/voice/whisperModels.js); None = its default. + dictation_model: Optional[str] = None anthropic_api_key: Optional[str] = None browser_homepage: str = "https://www.google.com" openai_api_key: Optional[str] = None diff --git a/electron/main.js b/electron/main.js index efece4e3..56b22d0f 100644 --- a/electron/main.js +++ b/electron/main.js @@ -1,5 +1,6 @@ const { app, components, BrowserWindow, ipcMain, shell, session, dialog, crashReporter, powerMonitor, Menu, clipboard, globalShortcut } = require('electron'); const whisperService = require('./voice/whisperService'); +const whisperModels = require('./voice/whisperModels'); const { injectText } = require('./voice/textInjector'); // Browser cards live in their own persistent partition so cookies/localStorage/IndexedDB survive reload + quit (Discord etc. stay logged in) and site data stays isolated from the app's defaultSession. The "clear browsing data" wipe nukes only this partition. MUST match BROWSER_PARTITION in frontend BrowserCard.tsx. @@ -2932,6 +2933,17 @@ ipcMain.handle('voice:warmup', async () => { }); // First-run model download progress so the pill can show "Preparing voice N%". ipcMain.handle('voice:status', () => whisperService.modelStatus()); +// Settings' model picker: the catalog with per-model install state, and the user's pick. +ipcMain.handle('voice:models', () => ({ + models: whisperModels.catalog(voiceUserDataDir()), + selected: whisperService.selectedModel(), +})); +ipcMain.handle('voice:set-model', (_e, id) => { + const ready = whisperService.setModel(voiceUserDataDir(), String(id || '')); + // Already on disk: warm the new one now so the next phrase is instant, same as at boot. + if (ready) whisperService.warmInBackground(voiceResourceDir(), voiceUserDataDir()); + return { ok: true, ready }; +}); // Paste the text into the frontmost app (dictate-anywhere). Returns whether the OS paste actually fired. ipcMain.handle('voice:inject', async (_e, text) => { try { const pasted = await injectText(String(text || '')); return { ok: true, pasted }; } catch (err) { return { ok: false, error: String(err && err.message ? err.message : err) }; } diff --git a/electron/preload.js b/electron/preload.js index a508f3ab..98be66a3 100644 --- a/electron/preload.js +++ b/electron/preload.js @@ -64,6 +64,9 @@ contextBridge.exposeInMainWorld('openswarm', { // text into the frontmost app; warmup pre-loads the model; onVoiceToggle fires on the global hotkey. voiceWarmup: () => ipcRenderer.invoke('voice:warmup'), voiceStatus: () => ipcRenderer.invoke('voice:status'), + // Settings' model picker: the catalog with install state, and switching (downloads on demand). + voiceModels: () => ipcRenderer.invoke('voice:models'), + voiceSetModel: (id) => ipcRenderer.invoke('voice:set-model', id), voiceTranscribe: (wavArrayBuffer) => ipcRenderer.invoke('voice:transcribe', wavArrayBuffer), voiceInject: (text) => ipcRenderer.invoke('voice:inject', text), onVoiceToggle: (cb) => { diff --git a/electron/voice/whisperModels.js b/electron/voice/whisperModels.js new file mode 100644 index 00000000..d2df278b --- /dev/null +++ b/electron/voice/whisperModels.js @@ -0,0 +1,149 @@ +// The dictation model catalog: which whisper builds we offer, where they come from, and how a +// chosen one gets onto disk intact. Split out of whisperService.js because "which file" and "the +// running server" are different jobs with different failure modes. +// +// Only English models: dictation lands in the user's own text field, and the .en builds are both +// smaller and more accurate than their multilingual siblings at the same size. Quantized (q5_1) +// variants are the field default (Handy and FluidVoice both standardised on quantized weights); +// they trade a sliver of accuracy for a third of the download. +// +// sha256 is the trust anchor. These are the HuggingFace LFS object ids, which ARE the file digests, +// so a truncated, resumed-wrong, or proxy-substituted body is rejected instead of handed to a +// native parser. + +const fs = require('fs'); +const path = require('path'); +const https = require('https'); +const crypto = require('crypto'); + +const MODELS = [ + { id: 'tiny.en-q5_1', file: 'ggml-tiny.en-q5_1.bin', label: 'Tiny', note: 'Fastest, least accurate', bytes: 32166155, sha256: 'c77c5766f1cef09b6b7d47f21b546cbddd4157886b3b5d6d4f709e91e66c7c2b' }, + { id: 'base.en-q5_1', file: 'ggml-base.en-q5_1.bin', label: 'Base (compact)', note: 'Base accuracy at a third the size', bytes: 59721011, sha256: '4baf70dd0d7c4247ba2b81fafd9c01005ac77c2f9ef064e00dcf195d0e2fdd2f' }, + { id: 'base.en', file: 'ggml-base.en.bin', label: 'Base', note: 'Balanced, the default', bytes: 147964211, sha256: 'a03779c86df3323075f5e796cb2ce5029f00ec8869eee3fdfb897afe36c6d002' }, + { id: 'small.en-q5_1', file: 'ggml-small.en-q5_1.bin', label: 'Small', note: 'Most accurate, slowest', bytes: 190098681, sha256: 'bfdff4894dcb76bbf647d56263ea2a96645423f1669176f4844a1bf8e478ad30' }, +]; + +const DEFAULT_MODEL_ID = 'base.en'; +const BASE_URL = 'https://huggingface.co/ggerganov/whisper.cpp/resolve/main/'; + +function modelById(id) { + return MODELS.find((m) => m.id === id) || MODELS.find((m) => m.id === DEFAULT_MODEL_ID); +} + +function modelDir(userDataDir) { + return path.join(userDataDir, 'whisper'); +} + +function modelPath(userDataDir, id) { + return path.join(modelDir(userDataDir), modelById(id).file); +} + +// A file is only "installed" once it is exactly the size the catalog says. A short file is a +// half-download someone's disk filled up on, and whisper would die parsing it. +function isInstalled(userDataDir, id) { + const m = modelById(id); + try { return fs.statSync(modelPath(userDataDir, id)).size === m.bytes; } catch (_) { return false; } +} + +// Resolve the model file to actually load: env override, then the user's pick, then anything else +// already on disk, then bundled. Never returns a path that isn't a complete file. +function resolveModelFile(resourceDir, userDataDir, id) { + if (process.env.OPENSWARM_WHISPER_MODEL && fs.existsSync(process.env.OPENSWARM_WHISPER_MODEL)) { + return process.env.OPENSWARM_WHISPER_MODEL; + } + if (id && isInstalled(userDataDir, id)) return modelPath(userDataDir, id); + const bundled = path.join(resourceDir, modelById(DEFAULT_MODEL_ID).file); + if (fs.existsSync(bundled)) return bundled; + // The user's pick isn't down yet but something else is: dictate with what we have rather than + // dead-ending on "no model" while a 190MB download runs. + const fallback = MODELS.find((m) => isInstalled(userDataDir, m.id)); + return fallback ? modelPath(userDataDir, fallback.id) : null; +} + +const download = { active: false, id: null, pct: 0, error: null }; + +function downloadStatus() { + return { downloading: download.active, id: download.id, pct: download.pct, error: download.error }; +} + +function sha256File(file) { + return new Promise((resolve, reject) => { + const h = crypto.createHash('sha256'); + const s = fs.createReadStream(file); + s.on('data', (c) => h.update(c)); + s.on('end', () => resolve(h.digest('hex'))); + s.on('error', reject); + }); +} + +// Fetch a catalog model into the user data dir. One at a time, resumable-free but atomic: bytes land +// in a .part file that only gets its real name after the digest matches. +function downloadModel(userDataDir, id) { + if (download.active) return; + const m = modelById(id); + const dest = modelPath(userDataDir, m.id); + download.active = true; + download.id = m.id; + download.pct = 0; + download.error = null; + try { fs.mkdirSync(modelDir(userDataDir), { recursive: true }); } catch (_) {} + const tmp = `${dest}.part`; + try { fs.unlinkSync(tmp); } catch (_) {} + + const fail = (msg) => { + download.active = false; + download.error = String(msg); + try { fs.unlinkSync(tmp); } catch (_) {} + }; + + // HuggingFace bounces resolve -> CDN -> signed URL, so follow redirects instead of assuming one hop. + const fetchUrl = (url, hops) => { + if (hops > 6) { fail('too-many-redirects'); return; } + const req = https.get(url, { headers: { 'User-Agent': 'openswarm-voice' } }, (res) => { + const code = res.statusCode || 0; + if (code >= 300 && code < 400 && res.headers.location) { + res.resume(); // drain so the socket frees + fetchUrl(new URL(res.headers.location, url).toString(), hops + 1); + return; + } + if (code !== 200) { res.resume(); fail(`http-${code}`); return; } + let got = 0; + const file = fs.createWriteStream(tmp); + res.on('data', (c) => { + got += c.length; + download.pct = Math.min(99, Math.round((got / m.bytes) * 100)); + // A server that keeps sending past the advertised size must not be allowed to fill the disk. + if (got > m.bytes) { req.destroy(); fail('oversize'); } + }); + res.pipe(file); + file.on('finish', () => file.close(async () => { + if (!download.active) return; // already failed out + if (got !== m.bytes) { fail('truncated'); return; } + try { + const digest = await sha256File(tmp); + if (digest !== m.sha256) { fail('checksum-mismatch'); return; } + fs.renameSync(tmp, dest); + download.pct = 100; + download.active = false; + } catch (e) { fail(e && e.message ? e.message : e); } + })); + res.on('error', () => fail('stream-error')); + file.on('error', () => fail('write-error')); + }); + req.on('error', (e) => fail(e && e.message ? e.message : e)); + }; + fetchUrl(`${BASE_URL}${m.file}`, 0); +} + +// What Settings renders: every model, its size, and whether it is already on this machine. +function catalog(userDataDir) { + return MODELS.map((m) => ({ + id: m.id, + label: m.label, + note: m.note, + sizeMb: Math.round(m.bytes / 1048576), + installed: isInstalled(userDataDir, m.id), + })); +} + +module.exports = { MODELS, DEFAULT_MODEL_ID, catalog, downloadModel, downloadStatus, isInstalled, modelById, modelPath, resolveModelFile }; diff --git a/electron/voice/whisperService.js b/electron/voice/whisperService.js index b48fd697..809c7103 100644 --- a/electron/voice/whisperService.js +++ b/electron/voice/whisperService.js @@ -6,59 +6,15 @@ const { spawn } = require('child_process'); const path = require('path'); const fs = require('fs'); -const https = require('https'); const net = require('net'); const os = require('os'); +const whisperModels = require('./whisperModels'); -const MODEL_FILE = 'ggml-base.en.bin'; -const MODEL_URL = 'https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en.bin'; - -// First-run model fetch, so a dev build (or a prod build that shipped without the model) still works -// instead of dead-ending on "no model". Progress is exposed so the pill can say "Preparing voice 40%". -const download = { active: false, pct: 0, error: null }; - -function downloadModel(dest) { - if (download.active) return; - download.active = true; - download.pct = 0; - download.error = null; - try { fs.mkdirSync(path.dirname(dest), { recursive: true }); } catch (_) {} - const tmp = `${dest}.part`; - try { fs.unlinkSync(tmp); } catch (_) {} - - const fail = (msg) => { download.active = false; download.error = String(msg); try { fs.unlinkSync(tmp); } catch (_) {} }; - - // HuggingFace bounces resolve -> CDN -> signed URL, so follow redirects instead of assuming one hop. - const fetchUrl = (url, hops) => { - if (hops > 6) { fail('too-many-redirects'); return; } - const req = https.get(url, { headers: { 'User-Agent': 'openswarm-voice' } }, (res) => { - const code = res.statusCode || 0; - if (code >= 300 && code < 400 && res.headers.location) { - res.resume(); // drain so the socket frees - fetchUrl(new URL(res.headers.location, url).toString(), hops + 1); - return; - } - if (code !== 200) { res.resume(); fail(`http-${code}`); return; } - const total = Number(res.headers['content-length'] || 0); - let got = 0; - const file = fs.createWriteStream(tmp); - res.on('data', (c) => { got += c.length; if (total) download.pct = Math.round((got / total) * 100); }); - res.pipe(file); - file.on('finish', () => file.close(() => { - // A truncated download is worse than none: only accept a complete file. - if (total && got < total) { fail('truncated'); return; } - try { fs.renameSync(tmp, dest); download.pct = 100; download.active = false; } catch (e) { fail(e && e.message ? e.message : e); } - })); - res.on('error', () => fail('stream-error')); - file.on('error', () => fail('write-error')); - }); - req.on('error', (e) => fail(e && e.message ? e.message : e)); - }; - fetchUrl(MODEL_URL, 0); -} +// Which catalog model the user picked. Settings pushes it in; until then the catalog default wins. +let selectedModelId = whisperModels.DEFAULT_MODEL_ID; function modelStatus() { - return { downloading: download.active, pct: download.pct, error: download.error }; + return whisperModels.downloadStatus(); } // Resolve the whisper-server binary. Env override wins (dev convenience), then the bundled per-arch @@ -75,22 +31,15 @@ function resolveBinary(resourceDir) { return exe; // last resort: hope it is on PATH } -// Resolve the model file. Env override, then bundled, then a dev cache under the app's data dir. function resolveModel(resourceDir, userDataDir) { - if (process.env.OPENSWARM_WHISPER_MODEL && fs.existsSync(process.env.OPENSWARM_WHISPER_MODEL)) { - return process.env.OPENSWARM_WHISPER_MODEL; - } - const bundled = path.join(resourceDir, MODEL_FILE); - if (fs.existsSync(bundled)) return bundled; - const cached = path.join(userDataDir, 'whisper', MODEL_FILE); - if (fs.existsSync(cached)) return cached; - return null; + return whisperModels.resolveModelFile(resourceDir, userDataDir, selectedModelId); } let proc = null; let port = 0; let readyPromise = null; let idleTimer = null; +let loadedModelFile = null; // A warm server holds ~210MB (measured, base.en). Free it after a long quiet spell; the reload that // costs is the FIRST one on a cold page cache, and boot-warm already paid that. @@ -152,28 +101,55 @@ async function p_primeGraph(p) { } catch (_) { /* priming is an optimization; a failure just means the first phrase pays it */ } } +// Pull the one line a human can act on out of whisper's chatty output. +function p_reasonFrom(tail) { + const lines = tail.split('\n').map((l) => l.trim()).filter(Boolean); + const blame = lines.filter((l) => /error|failed|invalid|unable|cannot|no such|not found/i.test(l)); + const pick = blame.length ? blame[blame.length - 1] : lines[lines.length - 1]; + return pick ? `: ${pick.slice(0, 200)}` : ''; +} + async function p_bootServer(resourceDir, userDataDir) { const bin = resolveBinary(resourceDir); const model = resolveModel(resourceDir, userDataDir); if (!model) { // Kick off a one-time background fetch so the NEXT dictation just works. - downloadModel(path.join(userDataDir, 'whisper', MODEL_FILE)); - throw new Error(download.active ? 'model-downloading' : 'no-model'); + whisperModels.downloadModel(userDataDir, selectedModelId); + throw new Error(whisperModels.downloadStatus().downloading ? 'model-downloading' : 'no-model'); } + loadedModelFile = model; const p = await freePort(); // No --convert: our WAV is already 16kHz mono, and the flag makes whisper demand ffmpeg on PATH at boot; a Finder-launched app has no brew PATH, so it exited before ever binding the port. const child = spawn(bin, ['-m', model, '--port', String(p), '-nt'], { cwd: os.tmpdir(), // whisper writes per-request temp files beside cwd, and a Finder launch starts at the unwritable / - stdio: ['ignore', 'ignore', 'pipe'], + // BOTH pipes: whisper writes its fatal reasons to STDOUT and then exits 0, so an ignored stdout + // turns "ffmpeg is missing" into an unexplained failure. Draining also stops the pipe buffer + // filling and blocking the child. + stdio: ['ignore', 'pipe', 'pipe'], }); - // Drain stderr or the pipe buffer fills and blocks the child; it is also the only diagnostic we get. - child.stderr.on('data', (c) => console.log('[voice] whisper:', String(c).trim().slice(0, 400))); + // Keep a rolling tail rather than the last chunk: whisper prints its real reason and THEN keeps + // banner-dumping, so "the most recent bytes" is reliably the least useful line it wrote. + let tail = ''; + const drain = (c) => { + const said = String(c); + if (!said.trim()) return; + tail = (tail + said).slice(-4000); + console.log('[voice] whisper:', said.trim().slice(0, 400)); + }; + child.stdout.on('data', drain); + child.stderr.on('data', drain); child.on('error', () => { proc = null; port = 0; }); child.on('exit', (code) => { if (code) console.log(`[voice] whisper-server exited code=${code}`); proc = null; port = 0; readyPromise = null; }); // Cold model load measured 15-38s on an M2; the old 20s budget timed out real first uses. const ok = await waitForReady(child, p, 60000); if (!ok) { try { child.kill(); } catch (_) {} + // A dead child is not a slow one. Whisper can die in ~0.1s with exit code 0 (a missing ffmpeg on + // a Finder-launched PATH does exactly that), so report ITS reason instantly instead of making the + // user sit through the full ready budget for a process that was never coming back. + if (child.exitCode !== null || child.signalCode !== null) { + throw new Error(`whisper-exited-${child.exitCode}${p_reasonFrom(tail)}`); + } throw new Error('server-timeout'); } proc = child; @@ -190,6 +166,9 @@ async function p_bootServer(resourceDir, userDataDir) { // settled-rejected promise forever so every later call kept throwing "model-downloading" even after // the model finished. Clearing on rejection here lets the next call retry cleanly. async function ensureServer(resourceDir, userDataDir) { + // A warm server is only reusable if it holds the file we would load now: a model switch, or the + // user's pick finishing its download while a fallback was serving, has to re-boot. + if (proc && port && resolveModel(resourceDir, userDataDir) !== loadedModelFile) stopServer(); if (proc && port) return port; if (readyPromise) return readyPromise; readyPromise = p_bootServer(resourceDir, userDataDir); @@ -234,6 +213,22 @@ function stopServer() { proc = null; port = 0; readyPromise = null; + loadedModelFile = null; +} + +// Settings picked a different model. Downloads it if missing; the running server keeps serving the +// old one until the new file is complete, so switching never leaves dictation dead in between. +function setModel(userDataDir, id) { + const next = whisperModels.modelById(id).id; + if (next === selectedModelId) return whisperModels.isInstalled(userDataDir, next); + selectedModelId = next; + if (whisperModels.isInstalled(userDataDir, next)) return true; + whisperModels.downloadModel(userDataDir, next); + return false; +} + +function selectedModel() { + return selectedModelId; } function isWarm() { @@ -248,4 +243,4 @@ async function reprimeAfterWake() { return true; } -module.exports = { ensureServer, warmInBackground, reprimeAfterWake, transcribe, stopServer, isWarm, resolveBinary, resolveModel, modelStatus }; +module.exports = { ensureServer, warmInBackground, reprimeAfterWake, transcribe, stopServer, isWarm, setModel, selectedModel, resolveBinary, resolveModel, modelStatus }; diff --git a/frontend/src/app/pages/Settings/sections/general/DictationModelPicker.tsx b/frontend/src/app/pages/Settings/sections/general/DictationModelPicker.tsx new file mode 100644 index 00000000..35bd4150 --- /dev/null +++ b/frontend/src/app/pages/Settings/sections/general/DictationModelPicker.tsx @@ -0,0 +1,97 @@ +import React, { useCallback, useEffect, useState } from 'react'; +import Box from '@mui/material/Box'; +import MenuItem from '@mui/material/MenuItem'; +import Select from '@mui/material/Select'; +import LinearProgress from '@mui/material/LinearProgress'; +import Typography from '@mui/material/Typography'; +import type { VoiceModel } from '@/types/electron'; +import { useClaudeTokens } from '@/shared/styles/ThemeContext'; + +// The speech model is the biggest speed/accuracy lever dictation has, so it is a real choice rather +// than a hidden constant. Picking one that isn't downloaded yet starts the fetch and shows progress; +// dictation keeps working on whatever model is already on disk until the new one lands. +const DictationModelPicker: React.FC<{ + value: string | null; + onChange: (id: string) => void; +}> = ({ value, onChange }) => { + const c = useClaudeTokens(); + const [models, setModels] = useState([]); + const [selected, setSelected] = useState(''); + const [pct, setPct] = useState(0); + const [downloadingId, setDownloadingId] = useState(null); + const [error, setError] = useState(null); + + const refresh = useCallback(async (): Promise => { + const res = await window.openswarm?.voiceModels?.(); + if (!res) return; + setModels(res.models); + setSelected(value || res.selected); + }, [value]); + + useEffect(() => { void refresh(); }, [refresh]); + + // Poll only while a download is actually running, so an idle Settings tab costs nothing. + useEffect(() => { + if (!downloadingId) return undefined; + let live = true; + const tick = async (): Promise => { + const st = await window.openswarm?.voiceStatus?.(); + if (!live) return; + if (!st) { setDownloadingId(null); return; } + setPct(st.pct || 0); + if (st.error) { setError(st.error); setDownloadingId(null); return; } + if (!st.downloading) { setDownloadingId(null); void refresh(); return; } + window.setTimeout(() => { void tick(); }, 700); + }; + void tick(); + return () => { live = false; }; + }, [downloadingId, refresh]); + + const pick = useCallback(async (id: string): Promise => { + setSelected(id); + setError(null); + onChange(id); + const res = await window.openswarm?.voiceSetModel?.(id); + if (res && !res.ready) { setPct(0); setDownloadingId(id); } + else void refresh(); + }, [onChange, refresh]); + + if (!models.length) return null; // web build or no bridge: nothing to choose between + + return ( + + + {downloadingId && ( + + + + {`Downloading ${downloadingId}, ${pct}%`} + + + )} + {error && ( + + {`Download failed (${error}). Pick it again to retry.`} + + )} + + ); +}; + +export default DictationModelPicker; diff --git a/frontend/src/app/pages/Settings/sections/general/GeneralInterface.tsx b/frontend/src/app/pages/Settings/sections/general/GeneralInterface.tsx index c66ea94e..9cc29ddc 100644 --- a/frontend/src/app/pages/Settings/sections/general/GeneralInterface.tsx +++ b/frontend/src/app/pages/Settings/sections/general/GeneralInterface.tsx @@ -15,6 +15,7 @@ import AccentColorPad from '@/app/components/theme/AccentColorPad'; import type { SettingsStyles } from '../settingsStyles'; import { settingSelectAttrs } from '../settingSelect'; import ShortcutRecorderChip, { dictationDefaultCombo, comboDisplay } from './ShortcutRecorderChip'; +import DictationModelPicker from './DictationModelPicker'; const GeneralInterface: React.FC<{ form: AppSettings; @@ -104,6 +105,17 @@ const GeneralInterface: React.FC<{ /> + + + Dictation model + Runs on this machine, nothing is uploaded. Bigger is more accurate but slower to transcribe. + + setForm({ ...form, dictation_model: id })} + /> + + Accent color diff --git a/frontend/src/shared/state/settingsSlice.ts b/frontend/src/shared/state/settingsSlice.ts index 0b86d57e..7fb32645 100644 --- a/frontend/src/shared/state/settingsSlice.ts +++ b/frontend/src/shared/state/settingsSlice.ts @@ -46,6 +46,7 @@ export interface AppSettings { theme: 'light' | 'dark'; new_agent_shortcut: string; dictation_shortcut?: string | null; + dictation_model?: string | null; anthropic_api_key: string | null; openai_api_key?: string | null; google_api_key?: string | null; @@ -167,6 +168,7 @@ export const DEFAULT_SETTINGS: AppSettings = { theme: 'light', new_agent_shortcut: 'Meta+l', dictation_shortcut: null, + dictation_model: null, anthropic_api_key: null, browser_homepage: 'https://duckduckgo.com', auto_select_mode_on_new_agent: false, diff --git a/frontend/src/shared/voice/VoiceDictationContext.tsx b/frontend/src/shared/voice/VoiceDictationContext.tsx index d6a38ab4..06c216e1 100644 --- a/frontend/src/shared/voice/VoiceDictationContext.tsx +++ b/frontend/src/shared/voice/VoiceDictationContext.tsx @@ -12,12 +12,19 @@ export function VoiceDictationProvider({ children }: { children: React.ReactNode const { state, lastText, error, pct, feedback, toggle, start, stop, volumeRef } = useVoiceDictation(); const holdMode = useAppSelector((s) => s.settings.data.voice_hold_to_talk ?? true); const dictationShortcut = useAppSelector((s) => s.settings.data.dictation_shortcut ?? null); + const dictationModel = useAppSelector((s) => s.settings.data.dictation_model ?? null); // Push the user's combo to main on boot and on change so every hotkey tier rebinds live. useEffect(() => { const bridge = window as unknown as { openswarm?: { setVoiceHotkey?: (combo: string | null) => void } }; bridge.openswarm?.setVoiceHotkey?.(dictationShortcut); }, [dictationShortcut]); + + // Main boots with the catalog default, so a user who picked something else has to say so on every + // launch or dictation quietly runs on the wrong model. + useEffect(() => { + if (dictationModel) void window.openswarm?.voiceSetModel?.(dictationModel); + }, [dictationModel]); const stateRef = useRef(state); stateRef.current = state; const heldRef = useRef(false); diff --git a/frontend/src/types/electron.d.ts b/frontend/src/types/electron.d.ts index 55b62ef6..4d2a839c 100644 --- a/frontend/src/types/electron.d.ts +++ b/frontend/src/types/electron.d.ts @@ -1,5 +1,14 @@ export {}; +// One entry in the dictation model catalog, as the main process reports it. +export interface VoiceModel { + id: string; + label: string; + note: string; + sizeMb: number; + installed: boolean; +} + declare global { namespace JSX { interface IntrinsicElements { @@ -55,7 +64,9 @@ declare global { hardReset?: () => Promise; clearBrowserData?: () => Promise<{ ok: boolean }>; voiceWarmup?: () => Promise<{ ok: boolean; error?: string }>; - voiceStatus?: () => Promise<{ downloading: boolean; pct: number; error: string | null }>; + voiceStatus?: () => Promise<{ downloading: boolean; id: string | null; pct: number; error: string | null }>; + voiceModels?: () => Promise<{ models: VoiceModel[]; selected: string }>; + voiceSetModel?: (id: string) => Promise<{ ok: boolean; ready: boolean }>; voiceTranscribe?: (wav: ArrayBuffer) => Promise<{ ok: boolean; text?: string; error?: string }>; voiceInject?: (text: string) => Promise<{ ok: boolean; pasted?: boolean; error?: string }>; onVoiceToggle?: (cb: () => void) => () => void;