tts: drive sentence glow from exact server timestamps
Replace the fragile client-side RMS / Web Audio pause detection (which stalled on sentence 0 whenever the audio context was suspended or the MSE duration was unknown) with exact per-sentence timing from the speech service. For tts-1-f5, fetch via the new timestamps mode and light each sentence by comparing audio.currentTime to the returned start/end ms; falls back to proportional mapping if the rendered and spoken sentence counts differ. Cache now stores sentences for correct replay glow. Non-F5 models keep streaming playback with no glow.
This commit is contained in:
parent
0b66a50938
commit
34f4fc325b
1 changed files with 81 additions and 102 deletions
|
|
@ -962,27 +962,11 @@ async function fetchTTSStreaming(cleanText, model, voice) {
|
|||
};
|
||||
}
|
||||
|
||||
// ---- Per-sentence glow synced to TTS audio via Web Audio pause detection ----
|
||||
// Listens to the audio's RMS energy and advances the highlighted sentence when
|
||||
// a long inter-sentence silence is heard. Shorter (comma/clause) pauses are
|
||||
// detected too, reserved for a future word-level "bouncing ball" glow. Falls
|
||||
// back to char-proportional timing when Web Audio is unavailable/suspended.
|
||||
// Tune these live from the browser console (window.GLOW):
|
||||
window.GLOW = window.GLOW || {
|
||||
silenceRms: 0.015, // RMS below this counts as silence
|
||||
sentencePauseMs: 280, // silence at least this long advances one sentence
|
||||
commaPauseMs: 90, // shorter pauses (clause/comma) — detected, not yet shown
|
||||
fftSize: 1024
|
||||
};
|
||||
let _glowAudioCtx = null;
|
||||
function _getGlowCtx() {
|
||||
if (!_glowAudioCtx) {
|
||||
try { _glowAudioCtx = new (window.AudioContext || window.webkitAudioContext)(); }
|
||||
catch (e) { return null; }
|
||||
}
|
||||
if (_glowAudioCtx.state === 'suspended') { _glowAudioCtx.resume().catch(function () {}); }
|
||||
return _glowAudioCtx;
|
||||
}
|
||||
// ---- Per-sentence glow synced to TTS audio via exact server timestamps ----
|
||||
// The speech service (tts-1-f5) returns each sentence's start/end in ms — it
|
||||
// synthesizes one audio chunk per sentence, so the timing is exact, not guessed.
|
||||
// We wrap the rendered message into sentence spans and light the active one by
|
||||
// comparing audio.currentTime to those boundaries. No Web Audio / RMS heuristics.
|
||||
|
||||
function wrapSentencesForGlow(container) {
|
||||
if (!container || container.dataset.glowWrapped === '1') return;
|
||||
|
|
@ -994,12 +978,10 @@ function wrapSentencesForGlow(container) {
|
|||
const full = nodes.map(function (x) { return x.nodeValue; }).join('');
|
||||
const parts = full.match(/[^.!?]+[.!?]*\s*/g) || [full];
|
||||
const ranges = [];
|
||||
const weights = [];
|
||||
let start = 0;
|
||||
for (let i = 0; i < parts.length; i++) {
|
||||
ranges.push([start, start + parts[i].length]);
|
||||
start += parts[i].length;
|
||||
weights.push(parts[i].replace(/\s+/g, ' ').trim().length || 1);
|
||||
}
|
||||
function sentAt(pos) {
|
||||
for (let s = 0; s < ranges.length; s++) { if (pos < ranges[s][1]) return s; }
|
||||
|
|
@ -1026,19 +1008,23 @@ function wrapSentencesForGlow(container) {
|
|||
});
|
||||
container.dataset.glowWrapped = '1';
|
||||
container._glowSentenceCount = parts.length;
|
||||
container._glowWeights = weights;
|
||||
}
|
||||
|
||||
function attachSentenceGlow(audio, playButton) {
|
||||
// Light the spoken sentence using exact server timing. `sentences` is the
|
||||
// array from the speech service: [{index, text, start_ms, end_ms}, ...].
|
||||
// Without it (non-F5 models), we no-op rather than guess.
|
||||
function attachSentenceGlow(audio, playButton, sentences) {
|
||||
if (!audio || !playButton) return;
|
||||
if (!Array.isArray(sentences) || !sentences.length) return;
|
||||
const wrapper = playButton.closest('.message-wrapper');
|
||||
const container = wrapper && wrapper.querySelector('.message-content');
|
||||
if (!container) return;
|
||||
wrapSentencesForGlow(container);
|
||||
const spans = container.querySelectorAll('.tts-sentence');
|
||||
const count = container._glowSentenceCount || 0;
|
||||
if (!spans.length || count <= 0) return;
|
||||
const domCount = container._glowSentenceCount || 0;
|
||||
if (!spans.length || domCount <= 0) return;
|
||||
|
||||
const serverCount = sentences.length;
|
||||
let active = -1;
|
||||
function setActive(idx) {
|
||||
if (idx === active) return;
|
||||
|
|
@ -1048,68 +1034,57 @@ function attachSentenceGlow(audio, playButton) {
|
|||
});
|
||||
}
|
||||
function clear() { spans.forEach(function (s) { s.classList.remove('tts-reading'); }); }
|
||||
setActive(0);
|
||||
|
||||
// Char-proportional fallback boundaries.
|
||||
const weights = container._glowWeights || [];
|
||||
const total = weights.reduce(function (a, b) { return a + b; }, 0) || 1;
|
||||
const cum = [];
|
||||
let acc = 0;
|
||||
for (let i = 0; i < weights.length; i++) { acc += weights[i]; cum.push(acc / total); }
|
||||
|
||||
// Web Audio analyser — only route through it when the context is running,
|
||||
// so we never mute playback on a suspended (autoplay-blocked) context.
|
||||
let analyser = null;
|
||||
let data = null;
|
||||
const ctx = _getGlowCtx();
|
||||
if (ctx && ctx.state === 'running' && !audio._glowSourced) {
|
||||
try {
|
||||
const src = ctx.createMediaElementSource(audio);
|
||||
analyser = ctx.createAnalyser();
|
||||
analyser.fftSize = window.GLOW.fftSize;
|
||||
src.connect(analyser);
|
||||
analyser.connect(ctx.destination);
|
||||
data = new Float32Array(analyser.fftSize);
|
||||
audio._glowSourced = true;
|
||||
} catch (e) { analyser = null; }
|
||||
// The rendered message and the synthesized text usually split into the same
|
||||
// sentence count; when they don't, map server index onto DOM spans by ratio.
|
||||
function domIndexFor(serverIdx) {
|
||||
if (domCount === serverCount) return serverIdx;
|
||||
return Math.min(domCount - 1, Math.floor(serverIdx * domCount / serverCount));
|
||||
}
|
||||
|
||||
let inSilence = false;
|
||||
let silenceStart = 0;
|
||||
function rms() {
|
||||
analyser.getFloatTimeDomainData(data);
|
||||
let sum = 0;
|
||||
for (let k = 0; k < data.length; k++) sum += data[k] * data[k];
|
||||
return Math.sqrt(sum / data.length);
|
||||
}
|
||||
function tick() {
|
||||
if (audio.paused || audio.ended) return;
|
||||
const ct = audio.currentTime;
|
||||
if (analyser) {
|
||||
const e = rms();
|
||||
if (e < window.GLOW.silenceRms) {
|
||||
if (!inSilence) { inSilence = true; silenceStart = ct; }
|
||||
} else if (inSilence) {
|
||||
const gapMs = (ct - silenceStart) * 1000;
|
||||
inSilence = false;
|
||||
if (gapMs >= window.GLOW.sentencePauseMs) {
|
||||
setActive(Math.min(active + 1, count - 1));
|
||||
}
|
||||
}
|
||||
} else if (isFinite(audio.duration) && audio.duration > 0) {
|
||||
const frac = ct / audio.duration;
|
||||
let idx = 0;
|
||||
while (idx < cum.length - 1 && frac > cum[idx]) idx++;
|
||||
setActive(idx);
|
||||
const ms = audio.currentTime * 1000;
|
||||
let si = 0;
|
||||
for (let i = 0; i < serverCount; i++) {
|
||||
if (ms >= sentences[i].start_ms) si = i; else break;
|
||||
}
|
||||
setActive(domIndexFor(si));
|
||||
requestAnimationFrame(tick);
|
||||
}
|
||||
setActive(0);
|
||||
audio.addEventListener('play', function () { requestAnimationFrame(tick); });
|
||||
audio.addEventListener('ended', clear);
|
||||
audio.addEventListener('pause', function () { if (audio.ended) clear(); });
|
||||
if (!audio.paused) requestAnimationFrame(tick);
|
||||
}
|
||||
|
||||
// Fetch TTS with exact per-sentence timing (tts-1-f5 only).
|
||||
// Returns { audio, blob, blobUrl, sentences }; throws on a non-OK response.
|
||||
async function fetchTTSWithTimestamps(cleanText, model, voice) {
|
||||
const response = await fetch(TTS_API_URL, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
model: model,
|
||||
voice: voice,
|
||||
input: cleanText,
|
||||
response_format: "mp3",
|
||||
timestamps: true
|
||||
})
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`TTS timestamps request failed: ${response.status}`);
|
||||
}
|
||||
const data = await response.json();
|
||||
const bytes = Uint8Array.from(atob(data.audio), function (c) { return c.charCodeAt(0); });
|
||||
const blob = new Blob([bytes], { type: "audio/mpeg" });
|
||||
const blobUrl = URL.createObjectURL(blob);
|
||||
const audio = new Audio(blobUrl);
|
||||
audio.playbackRate = 0.9;
|
||||
return { audio, blob, blobUrl, sentences: data.sentences || [] };
|
||||
}
|
||||
|
||||
// Function to read text using TTS (for manual button clicks) - now with streaming
|
||||
async function speakText(text, playButton, messageId) {
|
||||
console.log("speakText called with:", {text: text, messageId, autoPlay: autoPlayTTS});
|
||||
|
|
@ -1178,6 +1153,20 @@ async function speakTextQueued(text, playButton, messageId) {
|
|||
const cacheKey = `${messageId}-${voiceSelectValue}`;
|
||||
const cleanText = text.replace(/[^a-zA-Z0-9\s.,!?]/g, '');
|
||||
|
||||
function bindLifecycle(audio) {
|
||||
currentQueuedAudio = audio;
|
||||
audio.onended = () => {
|
||||
console.log("TTS finished for:", messageId);
|
||||
currentQueuedAudio = null;
|
||||
resolve();
|
||||
};
|
||||
audio.onerror = () => {
|
||||
console.error("TTS audio error for:", messageId);
|
||||
currentQueuedAudio = null;
|
||||
reject(new Error("Audio playback failed"));
|
||||
};
|
||||
}
|
||||
|
||||
// Check if audio is cached
|
||||
if (audioCache[cacheKey]) {
|
||||
const cachedData = audioCache[cacheKey];
|
||||
|
|
@ -1186,40 +1175,30 @@ async function speakTextQueued(text, playButton, messageId) {
|
|||
const audio = new Audio(audioUrl);
|
||||
audio.playbackRate = 0.9;
|
||||
enableDownloadButton(messageId, playButton, audioUrl, voice);
|
||||
attachSentenceGlow(audio, playButton);
|
||||
|
||||
currentQueuedAudio = audio;
|
||||
audio.onended = () => {
|
||||
console.log("TTS finished for:", messageId);
|
||||
currentQueuedAudio = null;
|
||||
resolve();
|
||||
};
|
||||
audio.onerror = () => {
|
||||
console.error("TTS audio error for:", messageId);
|
||||
currentQueuedAudio = null;
|
||||
reject(new Error("Audio playback failed"));
|
||||
};
|
||||
attachSentenceGlow(audio, playButton, cachedData.sentences);
|
||||
bindLifecycle(audio);
|
||||
audio.play().catch(reject);
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
// Use streaming TTS for faster playback start
|
||||
// F5 returns exact per-sentence timing — fetch it whole so the glow
|
||||
// syncs to real boundaries. Trades streaming for accurate highlight.
|
||||
if (model === 'tts-1-f5') {
|
||||
const result = await fetchTTSWithTimestamps(cleanText, model, voice);
|
||||
const audio = result.audio;
|
||||
attachSentenceGlow(audio, playButton, result.sentences);
|
||||
bindLifecycle(audio);
|
||||
audioCache[cacheKey] = { blob: result.blob, sentences: result.sentences };
|
||||
enableDownloadButton(messageId, playButton, result.blobUrl, voice);
|
||||
audio.play().catch(reject);
|
||||
return;
|
||||
}
|
||||
|
||||
// Other models: streaming playback, no glow (no server timing).
|
||||
const result = await fetchTTSStreaming(cleanText, model, voice);
|
||||
const audio = result.audio;
|
||||
attachSentenceGlow(audio, playButton);
|
||||
|
||||
currentQueuedAudio = audio;
|
||||
audio.onended = () => {
|
||||
console.log("TTS finished for:", messageId);
|
||||
currentQueuedAudio = null;
|
||||
resolve();
|
||||
};
|
||||
audio.onerror = () => {
|
||||
console.error("TTS audio error for:", messageId);
|
||||
currentQueuedAudio = null;
|
||||
reject(new Error("Audio playback failed"));
|
||||
};
|
||||
bindLifecycle(audio);
|
||||
|
||||
// Handle streaming completion for caching
|
||||
if (result.streamed && result.streamingComplete) {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue