mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-11 04:07:44 +02:00
[eric] voice: AGC reverted (it clipped normal speech into garble), learned-noun bar raised and store versioned
This commit is contained in:
@@ -175,19 +175,10 @@ export function useVoiceDictation() {
|
||||
const endpointer = hold ? null : createSilenceDetector(ctx.sampleRate);
|
||||
const streamRes = await window.openswarm?.voiceStreamStart?.();
|
||||
const streaming = streamRes?.ok === true;
|
||||
// Whisper-mode: quiet speech gets a gentle, slow-moving boost (3x cap) so murmured dictation
|
||||
// still clears the decode gates; loud input passes through untouched, and the gain glides so
|
||||
// it can never pump. Applied to BOTH the streamed chunks and the stored clip.
|
||||
let agcGain = 1;
|
||||
// No AGC: boosting "quiet" speech clipped NORMAL speech into distortion and wrecked accuracy
|
||||
// (Eric's garbled-history report). Whisper handles real levels fine; a whisper-mode boost can
|
||||
// only come back as an opt-in with a proper peak limiter, never inline on the hot path.
|
||||
const capture = await createCaptureNode(ctx, (i16) => {
|
||||
let sumSq = 0;
|
||||
for (let i = 0; i < i16.length; i += 8) { const v = i16[i] / 0x8000; sumSq += v * v; }
|
||||
const rawRms = Math.sqrt(sumSq / Math.max(1, Math.floor(i16.length / 8)));
|
||||
const desired = rawRms > 0.004 && rawRms < 0.03 ? Math.min(3, 0.06 / rawRms) : 1;
|
||||
agcGain = agcGain * 0.9 + desired * 0.1;
|
||||
if (agcGain > 1.02) {
|
||||
for (let i = 0; i < i16.length; i++) i16[i] = Math.max(-32768, Math.min(32767, Math.round(i16[i] * agcGain)));
|
||||
}
|
||||
if (streaming) window.openswarm?.voiceStreamChunk?.(i16.buffer as ArrayBuffer);
|
||||
const data = new Float32Array(i16.length);
|
||||
for (let i = 0; i < i16.length; i++) data[i] = i16[i] / 0x8000;
|
||||
|
||||
@@ -2,9 +2,11 @@
|
||||
// LEARNED from their own dictations (a capitalized word that keeps showing up is a name worth
|
||||
// biasing toward). All local; main receives one merged comma list via voiceSetDictionary.
|
||||
|
||||
const LEARNED_KEY = 'osw-dictation-learned';
|
||||
// v2: the v1 store learned junk from garbled transcripts and fed it BACK into decoding (a
|
||||
// degradation loop); new key orphans it, and the merge bar is higher.
|
||||
const LEARNED_KEY = 'osw-dictation-learned-v2';
|
||||
const LEARNED_CAP = 40;
|
||||
const MERGE_TOP = 20;
|
||||
const MERGE_TOP = 12;
|
||||
|
||||
// Words that start sentences get capitalized for free; only mid-sentence capitals count as names.
|
||||
const NOUN_RE = /(?<![.!?]\s)(?<!^)\b([A-Z][a-zA-Z]{2,}(?:'s)?)\b/g;
|
||||
@@ -24,7 +26,7 @@ function readLearned(): Record<string, number> {
|
||||
function pushMerged(): void {
|
||||
const counts = readLearned();
|
||||
const learned = Object.entries(counts)
|
||||
.filter(([, n]) => n >= 2)
|
||||
.filter(([w, n]) => n >= 3 && w.length >= 4)
|
||||
.sort((a, b) => b[1] - a[1])
|
||||
.slice(0, MERGE_TOP)
|
||||
.map(([w]) => w);
|
||||
|
||||
Reference in New Issue
Block a user