diff --git a/web/zebra-spaces.html b/web/zebra-spaces.html index c2d1608..d455c36 100644 --- a/web/zebra-spaces.html +++ b/web/zebra-spaces.html @@ -2093,29 +2093,110 @@ function loadWhisperCaptureWorklet(ctx){ }); } -let whisperPipeline = null; -let whisperLoading = false; -async function ensureWhisperPipeline(){ - if (whisperPipeline) return whisperPipeline; - if (whisperLoading){ - while (whisperLoading && !whisperPipeline) await new Promise(r => setTimeout(r, 200)); - return whisperPipeline; +/* Whisper inference runs in a dedicated Web Worker so ONNX Runtime + * never blocks the main thread. Without this, the page hangs for + * 1–3 seconds per chunk on a phone — visible as delayed join/leave + * chimes, button presses, scroll, etc. Fox 2026-06-05: "even the + * tones for entering and leaving are showing up way way later + * even on the host side." */ +const WHISPER_WORKER_CODE = ` +let transcriber = null; +let loading = false; +async function ensure(){ + if (transcriber) return transcriber; + if (loading){ + while (loading && !transcriber) await new Promise(r => setTimeout(r, 200)); + return transcriber; } - whisperLoading = true; - logLine('', 'whisper: loading model (~40MB, first time only)…'); + loading = true; try { - /* transformers.js from jsDelivr — runs ONNX Runtime + whisper-tiny.en - * fully in-browser. Model is downloaded once and cached by the - * service worker / browser cache; subsequent toggles are instant. */ const tx = await import('https://cdn.jsdelivr.net/npm/@huggingface/transformers@3.0.0'); - whisperPipeline = await tx.pipeline('automatic-speech-recognition', 'Xenova/whisper-tiny.en'); - logLine('', 'whisper: model ready'); + transcriber = await tx.pipeline('automatic-speech-recognition', 'Xenova/whisper-tiny.en'); + self.postMessage({ kind: 'ready' }); } catch (e){ - logLine('err', 'whisper load failed: '+e.message); + self.postMessage({ kind: 'error', message: e.message }); } finally { - whisperLoading = false; + loading = false; } - return whisperPipeline; + return transcriber; +} +self.onmessage = async (e) => { + if (!e.data) return; + if (e.data.cmd === 'init'){ + ensure(); + return; + } + if (e.data.cmd === 'transcribe'){ + const t = await ensure(); + if (!t){ + self.postMessage({ kind: 'result', id: e.data.id, error: 'pipeline unavailable' }); + return; + } + try { + const result = await t(e.data.chunk, { sampling_rate: 16000 }); + self.postMessage({ kind: 'result', id: e.data.id, text: (result && result.text) || '' }); + } catch (err){ + self.postMessage({ kind: 'result', id: e.data.id, error: err.message }); + } + } +}; +`; + +let whisperWorker = null; +let whisperWorkerReady = false; +let whisperWorkerLoading = false; +const _whisperPending = new Map(); /* id → { resolve, reject } */ +let _whisperReqId = 0; + +function ensureWhisperWorker(){ + if (whisperWorker) return whisperWorker; + whisperWorkerLoading = true; + logLine('', 'whisper: loading model (~40MB, first time only)…'); + const blob = new Blob([WHISPER_WORKER_CODE], { type: 'application/javascript' }); + const url = URL.createObjectURL(blob); + try { + whisperWorker = new Worker(url, { type: 'module' }); + } catch(e){ + logLine('err', 'whisper worker create: '+e.message); + URL.revokeObjectURL(url); + whisperWorkerLoading = false; + return null; + } + whisperWorker.onmessage = (e) => { + if (!e.data) return; + if (e.data.kind === 'ready'){ + whisperWorkerReady = true; + whisperWorkerLoading = false; + logLine('', 'whisper: worker ready'); + } else if (e.data.kind === 'error'){ + whisperWorkerLoading = false; + logLine('err', 'whisper worker init: '+e.data.message); + } else if (e.data.kind === 'result'){ + const cb = _whisperPending.get(e.data.id); + if (cb){ + _whisperPending.delete(e.data.id); + if (e.data.error) cb.reject(new Error(e.data.error)); + else cb.resolve(e.data.text || ''); + } + } + }; + whisperWorker.onerror = (ev) => { + logLine('err', 'whisper worker error: '+(ev.message || 'unknown')); + }; + whisperWorker.postMessage({ cmd: 'init' }); + return whisperWorker; +} + +function transcribeViaWorker(chunk){ + const w = ensureWhisperWorker(); + if (!w) return Promise.reject(new Error('worker unavailable')); + return new Promise((resolve, reject) => { + const id = ++_whisperReqId; + _whisperPending.set(id, { resolve, reject }); + /* transfer the chunk buffer — main thread no longer needs it */ + try { w.postMessage({ cmd: 'transcribe', id, chunk }, [chunk.buffer]); } + catch(e){ _whisperPending.delete(id); reject(e); } + }); } let transcribeEnabled = false; @@ -2139,6 +2220,22 @@ const WHISPER_HALLUCINATIONS = new Set([ * the speaker has paused — looks like spam in the log. */ const _lastTranscriptByUuid = new Map(); +/* Global serialization for Whisper inference. ONNX Runtime in + * transformers.js runs on the main thread (in WASM) and can block UI + * paint for 1–3 seconds per chunk on a phone. With N talking speakers + * the queue grows faster than we can process and the page hangs. + * + * Rule: only one chunk is "in flight" globally at any moment. When a + * new chunk arrives while busy, DROP it — don't queue. Stale chunks + * are worse than missing chunks (the speaker has moved on by the time + * we'd transcribe). + * + * Phase 2 (deferred): move ONNX into a Web Worker so inference never + * touches the UI thread at all. */ +let _whisperBusy = false; +let _whisperDroppedChunks = 0; +let _whisperLastDropLogAt = 0; + function appendTranscriptLine(uuid, text){ const log = document.getElementById('transcript-log'); if (!log) return; @@ -2172,11 +2269,22 @@ async function startCaptureForUuid(uuid){ const capture = new AudioWorkletNode(audioCtx, 'whisper-capture'); capture.port.onmessage = async (e) => { if (!e.data || !e.data.chunk) return; - const pipeline = await ensureWhisperPipeline(); - if (!pipeline) return; + /* Inflight gate: only one transcription in flight at a time, + * globally. If a new chunk arrives while busy, DROP it — the + * worker is still chewing on something else and queueing + * would just stack stale audio. */ + if (_whisperBusy){ + _whisperDroppedChunks++; + const now = Date.now(); + if (now - _whisperLastDropLogAt > 30000){ + _whisperLastDropLogAt = now; + logLine('', 'whisper: dropped '+_whisperDroppedChunks+' chunks (worker saturated)'); + } + return; + } + _whisperBusy = true; try { - const result = await pipeline(e.data.chunk, { sampling_rate: 16000 }); - const txt = (result && result.text ? result.text : '').trim(); + const txt = (await transcribeViaWorker(e.data.chunk)).trim(); if (txt.length < 2) return; if (/^\[/.test(txt)) return; /* "[BLANK_AUDIO]" etc. */ /* normalize for hallucination match: lowercase, strip trailing @@ -2189,6 +2297,8 @@ async function startCaptureForUuid(uuid){ appendTranscriptLine(uuid, txt); } catch (err){ logLine('err', 'whisper transcribe '+uuid.slice(0,4)+': '+err.message); + } finally { + _whisperBusy = false; } }; /* tap the source — capture runs in parallel with the worklet @@ -2220,8 +2330,8 @@ async function toggleTranscribe(){ if (btn) btn.textContent = transcribeEnabled ? 'transcribe (on)' : 'transcribe (off)'; if (sec) sec.classList.toggle('hidden', !transcribeEnabled); if (transcribeEnabled){ - /* pre-warm the pipeline so the first chunk doesn't wait. */ - ensureWhisperPipeline().catch(()=>{}); + /* pre-warm the worker so the first chunk doesn't wait on model load. */ + ensureWhisperWorker(); for (const [uuid] of listenerAudioNodes){ startCaptureForUuid(uuid).catch(e => logLine('err', 'transcribe start '+uuid+': '+e.message)); } @@ -7160,8 +7270,8 @@ logLine('', 'ready — pick a handle, type a rendezvous code, enter the space');