diff --git a/electron/voice/streamingSession.js b/electron/voice/streamingSession.js index 0eb0f364..2d2aaae3 100644 --- a/electron/voice/streamingSession.js +++ b/electron/voice/streamingSession.js @@ -16,6 +16,12 @@ const SAMPLE_RATE = 16000; const SILENT_KEEP_BYTES = SAMPLE_RATE * 2 * 2; const SILENT_TRIM_BYTES = SAMPLE_RATE * 2 * 10; +// Whisper captions non-speech in brackets/parens ("[ Background sounds ]", "(laughs)"); strip them so +// neither the live preview nor a committed phrase ever carries a caption instead of dictation. +function stripSoundCaptions(text) { + return String(text || '').replace(/\[[^\]]*\]|\([^)]*\)|\*[^*]*\*|(?:^|\s)>>\s?/g, ' ').replace(/\s+/g, ' ').trim(); +} + function wavFromPcm16(pcm) { const buf = Buffer.alloc(44 + pcm.length); buf.write('RIFF', 0); buf.writeUInt32LE(36 + pcm.length, 4); buf.write('WAVE', 8); @@ -68,7 +74,7 @@ function createStreamingSession({ resourceDir, userDataDir, onPartial, previewIn inflight = decodePcm(pcm) .then((text) => { if (closedDown || epoch !== segEpoch) return; // the segment closed mid-decode; its final wins - tentative = text; + tentative = stripSoundCaptions(text); emit(); }) .catch(() => { skipNext = true; }) @@ -94,7 +100,7 @@ function createStreamingSession({ resourceDir, userDataDir, onPartial, previewIn commitChain = commitChain.then(async () => { if (inflight) await inflight; try { - const text = await decodePcm(pcm); + const text = stripSoundCaptions(await decodePcm(pcm)); if (text) committed.push(text); } catch (_) { degraded = true; // a lost phrase final means the caller must fall back to the full-clip decode @@ -137,4 +143,4 @@ function createStreamingSession({ resourceDir, userDataDir, onPartial, previewIn }; } -module.exports = { createStreamingSession, wavFromPcm16 }; +module.exports = { createStreamingSession, wavFromPcm16, stripSoundCaptions }; diff --git a/electron/voice/streamingVoice.test.js b/electron/voice/streamingVoice.test.js index 5aa24dcc..33eda9bd 100644 --- a/electron/voice/streamingVoice.test.js +++ b/electron/voice/streamingVoice.test.js @@ -2,7 +2,7 @@ const { test } = require('node:test'); const assert = require('node:assert'); const { createStreamSegmenter } = require('./streamSegmenter'); const whisperService = require('./whisperService'); -const { createStreamingSession, wavFromPcm16 } = require('./streamingSession'); +const { createStreamingSession, wavFromPcm16, stripSoundCaptions } = require('./streamingSession'); const RATE = 16000; @@ -91,6 +91,24 @@ test('session: a failed segment decode marks the result degraded', async () => { assert.strictEqual(out.degraded, true); }); +test('stripSoundCaptions: captions and speaker marks go, words stay', () => { + assert.strictEqual(stripSoundCaptions('[ Background sounds ]'), ''); + assert.strictEqual(stripSoundCaptions('[ Silence ] >> Hello world. [ Silence ]'), 'Hello world.'); + assert.strictEqual(stripSoundCaptions('(laughs) okay *music* done'), 'okay done'); + assert.strictEqual(stripSoundCaptions('plain dictated text'), 'plain dictated text'); +}); + +test('session: a caption-only decode never becomes a committed phrase', async () => { + const restore = stubTranscribe(async () => '[ Background sounds ]'); + const s = createStreamingSession({ resourceDir: '', userDataDir: '', onPartial: () => {}, previewIntervalMs: 3600000 }); + s.pushChunk(asBuffer(tone(400))); + const out = await s.stop(); + restore(); + assert.strictEqual(out.ok, true); + assert.strictEqual(out.text, ''); + assert.strictEqual(out.degraded, false); +}); + test('session: chunks after cancel are dropped and stop reports stopped', async () => { const restore = stubTranscribe(async () => 'never'); const s = createStreamingSession({ resourceDir: '', userDataDir: '', onPartial: () => {}, previewIntervalMs: 3600000 }); diff --git a/frontend/src/shared/voice/useVoiceDictation.ts b/frontend/src/shared/voice/useVoiceDictation.ts index 9b47f2d2..06f8b36c 100644 --- a/frontend/src/shared/voice/useVoiceDictation.ts +++ b/frontend/src/shared/voice/useVoiceDictation.ts @@ -200,6 +200,8 @@ export function useVoiceDictation() { res = await window.openswarm?.voiceTranscribe?.(wav); } } + // Whisper captions non-speech in brackets/parens ("[ Background sounds ]", "(laughs)") and marks speaker turns with ">>"; those are annotations, not dictation. + if (res?.ok && res.text) res = { ok: true, text: res.text.replace(/\[[^\]]*\]|\([^)]*\)|\*[^*]*\*|(?:^|\s)>>\s?/g, ' ').replace(/\s+/g, ' ').trim() }; // A transcript with no letter or digit in it is a hallucination artifact (the lone comma), not dictation. if (res?.ok && res.text && !/[\p{L}\p{N}]/u.test(res.text)) res = { ok: true, text: '' }; if (res?.ok && res.text) {