Files
openswarm/frontend/src/shared/voice/injectAtFocus.ts
T

63 lines
3.4 KiB
TypeScript

import { getLastInteractedBrowser } from '@/shared/browserFocus';
import { getWebview } from '@/shared/browserRegistry';
import { takeInjectSnapshot, setInjectSnapshot, isUsableTarget } from './injectTargetSnapshot';
// Dictation lands where the user's cursor actually is, like every real dictation tool: a focused
// in-app field gets the text typed in (undo-friendly, fires React input events), a focused browser
// card forwards into the guest page's field, anything else falls back to the OS-level paste.
export type InjectTarget = 'field' | 'webview' | 'composer' | null;
export function injectAtFocus(text: string): InjectTarget {
const snap = takeInjectSnapshot();
// The field the user dictated into is gone. Whatever holds focus now is a stranger's box, and
// typing their words into it is worse than dropping them, so drop them.
if (snap.targetLost && !snap.browserId) return null;
const active = snap.el || (document.activeElement as HTMLElement | null);
if (active && (active.tagName === 'INPUT' || active.tagName === 'TEXTAREA' || active.isContentEditable)) {
try {
active.focus();
// execCommand keeps the undo stack and fires the input events React listens for; the manual
// fallback covers fields where Chromium refuses the command (rare, e.g. type=number).
const ok = document.execCommand('insertText', false, text);
if (!ok && (active.tagName === 'INPUT' || active.tagName === 'TEXTAREA')) {
const el = active as HTMLInputElement | HTMLTextAreaElement;
const s = el.selectionStart ?? el.value.length;
const e = el.selectionEnd ?? el.value.length;
el.value = el.value.slice(0, s) + text + el.value.slice(e);
el.selectionStart = el.selectionEnd = s + text.length;
el.dispatchEvent(new Event('input', { bubbles: true }));
}
return 'field';
} catch {
return null;
}
}
// A webview steals focus when the user clicks into a page, so activeElement IS the webview tag.
const focusedTag = active && active.tagName === 'WEBVIEW' ? (active as unknown as { insertText?: (t: string) => Promise<void> }) : null;
if (focusedTag?.insertText) {
try { void focusedTag.insertText(text); return 'webview'; } catch { /* fall through */ }
}
// Last-interacted browser card: the user clicked a page field, then hit the hotkey.
const browserId = snap.browserId || getLastInteractedBrowser();
if (browserId) {
const wv = getWebview(browserId) as unknown as { insertText?: (t: string) => Promise<void>; focus?: () => void } | undefined;
if (wv?.insertText) {
try { wv.focus?.(); void wv.insertText(text); return 'webview'; } catch { /* fall through */ }
}
}
// No cursor anywhere: open the dashboard composer with the transcript typed in. Words are never dropped.
window.dispatchEvent(new CustomEvent('openswarm:dictation-fallback', { detail: { text } }));
return 'composer';
}
/** Called at press-start so the words land where the user was looking, not where focus drifted. */
export function snapshotInjectTarget(): void {
// Only a typeable element counts as "aimed at". document.activeElement is <body> when nothing is
// focused, and storing that would read as a lost target later and swallow the composer fallback.
const active = document.activeElement as HTMLElement | null;
setInjectSnapshot({
el: isUsableTarget(active) ? active : null,
browserId: getLastInteractedBrowser(),
});
}