tts: drive sentence glow from exact server timestamps

Replace the fragile client-side RMS / Web Audio pause detection (which
stalled on sentence 0 whenever the audio context was suspended or the
MSE duration was unknown) with exact per-sentence timing from the speech
service. For tts-1-f5, fetch via the new timestamps mode and light each
sentence by comparing audio.currentTime to the returned start/end ms;
falls back to proportional mapping if the rendered and spoken sentence
counts differ. Cache now stores sentences for correct replay glow.
Non-F5 models keep streaming playback with no glow.
This commit is contained in:
russell@unturf.com 2026-05-29 12:29:37 -04:00
parent 0b66a50938
commit 34f4fc325b
No known key found for this signature in database

View file

@ -962,27 +962,11 @@ async function fetchTTSStreaming(cleanText, model, voice) {
};
}
// ---- Per-sentence glow synced to TTS audio via Web Audio pause detection ----
// Listens to the audio's RMS energy and advances the highlighted sentence when
// a long inter-sentence silence is heard. Shorter (comma/clause) pauses are
// detected too, reserved for a future word-level "bouncing ball" glow. Falls
// back to char-proportional timing when Web Audio is unavailable/suspended.
// Tune these live from the browser console (window.GLOW):
window.GLOW = window.GLOW || {
silenceRms: 0.015, // RMS below this counts as silence
sentencePauseMs: 280, // silence at least this long advances one sentence
commaPauseMs: 90, // shorter pauses (clause/comma) — detected, not yet shown
fftSize: 1024
};
let _glowAudioCtx = null;
function _getGlowCtx() {
if (!_glowAudioCtx) {
try { _glowAudioCtx = new (window.AudioContext || window.webkitAudioContext)(); }
catch (e) { return null; }
}
if (_glowAudioCtx.state === 'suspended') { _glowAudioCtx.resume().catch(function () {}); }
return _glowAudioCtx;
}
// ---- Per-sentence glow synced to TTS audio via exact server timestamps ----
// The speech service (tts-1-f5) returns each sentence's start/end in ms — it
// synthesizes one audio chunk per sentence, so the timing is exact, not guessed.
// We wrap the rendered message into sentence spans and light the active one by
// comparing audio.currentTime to those boundaries. No Web Audio / RMS heuristics.
function wrapSentencesForGlow(container) {
if (!container || container.dataset.glowWrapped === '1') return;
@ -994,12 +978,10 @@ function wrapSentencesForGlow(container) {
const full = nodes.map(function (x) { return x.nodeValue; }).join('');
const parts = full.match(/[^.!?]+[.!?]*\s*/g) || [full];
const ranges = [];
const weights = [];
let start = 0;
for (let i = 0; i < parts.length; i++) {
ranges.push([start, start + parts[i].length]);
start += parts[i].length;
weights.push(parts[i].replace(/\s+/g, ' ').trim().length || 1);
}
function sentAt(pos) {
for (let s = 0; s < ranges.length; s++) { if (pos < ranges[s][1]) return s; }
@ -1026,19 +1008,23 @@ function wrapSentencesForGlow(container) {
});
container.dataset.glowWrapped = '1';
container._glowSentenceCount = parts.length;
container._glowWeights = weights;
}
function attachSentenceGlow(audio, playButton) {
// Light the spoken sentence using exact server timing. `sentences` is the
// array from the speech service: [{index, text, start_ms, end_ms}, ...].
// Without it (non-F5 models), we no-op rather than guess.
function attachSentenceGlow(audio, playButton, sentences) {
if (!audio || !playButton) return;
if (!Array.isArray(sentences) || !sentences.length) return;
const wrapper = playButton.closest('.message-wrapper');
const container = wrapper && wrapper.querySelector('.message-content');
if (!container) return;
wrapSentencesForGlow(container);
const spans = container.querySelectorAll('.tts-sentence');
const count = container._glowSentenceCount || 0;
if (!spans.length || count <= 0) return;
const domCount = container._glowSentenceCount || 0;
if (!spans.length || domCount <= 0) return;
const serverCount = sentences.length;
let active = -1;
function setActive(idx) {
if (idx === active) return;
@ -1048,68 +1034,57 @@ function attachSentenceGlow(audio, playButton) {
});
}
function clear() { spans.forEach(function (s) { s.classList.remove('tts-reading'); }); }
setActive(0);
// Char-proportional fallback boundaries.
const weights = container._glowWeights || [];
const total = weights.reduce(function (a, b) { return a + b; }, 0) || 1;
const cum = [];
let acc = 0;
for (let i = 0; i < weights.length; i++) { acc += weights[i]; cum.push(acc / total); }
// Web Audio analyser — only route through it when the context is running,
// so we never mute playback on a suspended (autoplay-blocked) context.
let analyser = null;
let data = null;
const ctx = _getGlowCtx();
if (ctx && ctx.state === 'running' && !audio._glowSourced) {
try {
const src = ctx.createMediaElementSource(audio);
analyser = ctx.createAnalyser();
analyser.fftSize = window.GLOW.fftSize;
src.connect(analyser);
analyser.connect(ctx.destination);
data = new Float32Array(analyser.fftSize);
audio._glowSourced = true;
} catch (e) { analyser = null; }
// The rendered message and the synthesized text usually split into the same
// sentence count; when they don't, map server index onto DOM spans by ratio.
function domIndexFor(serverIdx) {
if (domCount === serverCount) return serverIdx;
return Math.min(domCount - 1, Math.floor(serverIdx * domCount / serverCount));
}
let inSilence = false;
let silenceStart = 0;
function rms() {
analyser.getFloatTimeDomainData(data);
let sum = 0;
for (let k = 0; k < data.length; k++) sum += data[k] * data[k];
return Math.sqrt(sum / data.length);
}
function tick() {
if (audio.paused || audio.ended) return;
const ct = audio.currentTime;
if (analyser) {
const e = rms();
if (e < window.GLOW.silenceRms) {
if (!inSilence) { inSilence = true; silenceStart = ct; }
} else if (inSilence) {
const gapMs = (ct - silenceStart) * 1000;
inSilence = false;
if (gapMs >= window.GLOW.sentencePauseMs) {
setActive(Math.min(active + 1, count - 1));
}
}
} else if (isFinite(audio.duration) && audio.duration > 0) {
const frac = ct / audio.duration;
let idx = 0;
while (idx < cum.length - 1 && frac > cum[idx]) idx++;
setActive(idx);
const ms = audio.currentTime * 1000;
let si = 0;
for (let i = 0; i < serverCount; i++) {
if (ms >= sentences[i].start_ms) si = i; else break;
}
setActive(domIndexFor(si));
requestAnimationFrame(tick);
}
setActive(0);
audio.addEventListener('play', function () { requestAnimationFrame(tick); });
audio.addEventListener('ended', clear);
audio.addEventListener('pause', function () { if (audio.ended) clear(); });
if (!audio.paused) requestAnimationFrame(tick);
}
// Fetch TTS with exact per-sentence timing (tts-1-f5 only).
// Returns { audio, blob, blobUrl, sentences }; throws on a non-OK response.
async function fetchTTSWithTimestamps(cleanText, model, voice) {
const response = await fetch(TTS_API_URL, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
model: model,
voice: voice,
input: cleanText,
response_format: "mp3",
timestamps: true
})
});
if (!response.ok) {
throw new Error(`TTS timestamps request failed: ${response.status}`);
}
const data = await response.json();
const bytes = Uint8Array.from(atob(data.audio), function (c) { return c.charCodeAt(0); });
const blob = new Blob([bytes], { type: "audio/mpeg" });
const blobUrl = URL.createObjectURL(blob);
const audio = new Audio(blobUrl);
audio.playbackRate = 0.9;
return { audio, blob, blobUrl, sentences: data.sentences || [] };
}
// Function to read text using TTS (for manual button clicks) - now with streaming
async function speakText(text, playButton, messageId) {
console.log("speakText called with:", {text: text, messageId, autoPlay: autoPlayTTS});
@ -1178,6 +1153,20 @@ async function speakTextQueued(text, playButton, messageId) {
const cacheKey = `${messageId}-${voiceSelectValue}`;
const cleanText = text.replace(/[^a-zA-Z0-9\s.,!?]/g, '');
function bindLifecycle(audio) {
currentQueuedAudio = audio;
audio.onended = () => {
console.log("TTS finished for:", messageId);
currentQueuedAudio = null;
resolve();
};
audio.onerror = () => {
console.error("TTS audio error for:", messageId);
currentQueuedAudio = null;
reject(new Error("Audio playback failed"));
};
}
// Check if audio is cached
if (audioCache[cacheKey]) {
const cachedData = audioCache[cacheKey];
@ -1186,40 +1175,30 @@ async function speakTextQueued(text, playButton, messageId) {
const audio = new Audio(audioUrl);
audio.playbackRate = 0.9;
enableDownloadButton(messageId, playButton, audioUrl, voice);
attachSentenceGlow(audio, playButton);
currentQueuedAudio = audio;
audio.onended = () => {
console.log("TTS finished for:", messageId);
currentQueuedAudio = null;
resolve();
};
audio.onerror = () => {
console.error("TTS audio error for:", messageId);
currentQueuedAudio = null;
reject(new Error("Audio playback failed"));
};
attachSentenceGlow(audio, playButton, cachedData.sentences);
bindLifecycle(audio);
audio.play().catch(reject);
return;
}
try {
// Use streaming TTS for faster playback start
// F5 returns exact per-sentence timing — fetch it whole so the glow
// syncs to real boundaries. Trades streaming for accurate highlight.
if (model === 'tts-1-f5') {
const result = await fetchTTSWithTimestamps(cleanText, model, voice);
const audio = result.audio;
attachSentenceGlow(audio, playButton, result.sentences);
bindLifecycle(audio);
audioCache[cacheKey] = { blob: result.blob, sentences: result.sentences };
enableDownloadButton(messageId, playButton, result.blobUrl, voice);
audio.play().catch(reject);
return;
}
// Other models: streaming playback, no glow (no server timing).
const result = await fetchTTSStreaming(cleanText, model, voice);
const audio = result.audio;
attachSentenceGlow(audio, playButton);
currentQueuedAudio = audio;
audio.onended = () => {
console.log("TTS finished for:", messageId);
currentQueuedAudio = null;
resolve();
};
audio.onerror = () => {
console.error("TTS audio error for:", messageId);
currentQueuedAudio = null;
reject(new Error("Audio playback failed"));
};
bindLifecycle(audio);
// Handle streaming completion for caching
if (result.streamed && result.streamingComplete) {