From 19922ee9fadc1656d3966721f0134b9e02e3384e Mon Sep 17 00:00:00 2001 From: "russell@unturf.com" Date: Wed, 3 Jun 2026 15:15:49 -0400 Subject: [PATCH] =?UTF-8?q?zebra-report:=20deploy=20=E2=80=94=20mesh=20PC?= =?UTF-8?q?=20exponential=20backoff=20+=20give-up=20cap?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- zebra-report/zebra-spaces.html | 72 +++++++++++++++++++++++++++------- 1 file changed, 58 insertions(+), 14 deletions(-) diff --git a/zebra-report/zebra-spaces.html b/zebra-report/zebra-spaces.html index 15b9795..d32fe3f 100644 --- a/zebra-report/zebra-spaces.html +++ b/zebra-report/zebra-spaces.html @@ -3192,6 +3192,7 @@ async function handleSignal(raw){ handraise.delete(m.uuid); spotlights.delete(m.uuid); tearPeer(m.uuid); + clearMeshRetryState(m.uuid); if (hostUUID === m.uuid) hostUUID = ''; renderRoom(); reorderTiles(); @@ -3362,7 +3363,7 @@ async function handleSignal(raw){ removeVideoTile('gameshare', victPub); } catch(_){} } - members.delete(m.uuid); tearPeer(m.uuid); handraise.delete(m.uuid); + members.delete(m.uuid); tearPeer(m.uuid); clearMeshRetryState(m.uuid); handraise.delete(m.uuid); renderRoom(); } break; @@ -3518,8 +3519,30 @@ function updateRoleUI(){ * by tear + reconnect with the same rule, so the same side always * drives recovery. * ================================================================== */ +/* per-peer retry tracking — repeated mesh failures (NAT/firewall the + * page can't traverse, even with TURN) used to trigger a tight 1.5s + * reconnect loop forever. Each cycle ate CPU/network and added to + * the robot-voice/lost-mic noise we keep chasing. Now we exponential- + * backoff and after PEER_MESH_MAX_RETRIES we give up the mesh entirely + * for that peer and let SFU carry the audio. peer-left clears the + * tracking, so a fresh join from the same uuid starts a new budget. */ +const PEER_MESH_MAX_RETRIES = 4; +const peerMeshGiveUp = new Set(); // uuids we've stopped trying to mesh +const peerMeshRetries = new Map(); // uuid -> attempt count +const peerMeshTimers = new Map(); // uuid -> pending setTimeout id +function clearMeshRetryState(uuid){ + peerMeshGiveUp.delete(uuid); + peerMeshRetries.delete(uuid); + const t = peerMeshTimers.get(uuid); + if (t){ clearTimeout(t); peerMeshTimers.delete(uuid); } +} + async function connectToPeer(uuid, weOffer){ if (peers.has(uuid)) return; + if (peerMeshGiveUp.has(uuid)){ + logLine('', 'peer '+uuid+' mesh disabled (max retries) — staying on SFU'); + return; + } if (!micStream){ try { await getMic(); } catch(e){ logLine('err','mic for '+uuid+': '+e.message); return; } } const pc = new RTCPeerConnection(rtcConfig); peers.set(uuid, pc); @@ -3537,18 +3560,39 @@ async function connectToPeer(uuid, weOffer){ }; pc.onicecandidate = (ev) => { /* using waitForIceGathering pattern, candidates ignored */ }; pc.onconnectionstatechange = () => { - if (pc.connectionState === 'failed' && peers.get(uuid) === pc){ - logLine('', 'peer '+uuid+' failed — reconnecting'); - tearPeer(uuid); - /* Mesh PC just died; the audio element for this peer was bound to - * the dying mesh stream and won't recover on its own. Switch back - * to the cached SFU stream so the user keeps hearing them while - * mesh reconnect attempts run in the background. */ - try { attachCachedSfuStreamFor(uuid); } catch(_){} - /* let the offerer drive recovery */ - setTimeout(()=>{ if (members.has(uuid) && canSpeak(members.get(uuid).role) && canSpeak(myRole)) - connectToPeer(uuid, myUUID < uuid); }, 1500); + if (pc.connectionState === 'connected'){ + /* successful connect — reset the retry budget so a much-later + * transient failure gets a fresh round of attempts. */ + peerMeshRetries.delete(uuid); + return; } + if (pc.connectionState !== 'failed' || peers.get(uuid) !== pc) return; + + const attempts = (peerMeshRetries.get(uuid) || 0) + 1; + peerMeshRetries.set(uuid, attempts); + tearPeer(uuid); + /* Mesh PC just died; the audio element for this peer was bound to + * the dying mesh stream and won't recover on its own. Switch back + * to the cached SFU stream so the user keeps hearing them while + * mesh reconnect attempts run in the background. */ + try { attachCachedSfuStreamFor(uuid); } catch(_){} + + if (attempts >= PEER_MESH_MAX_RETRIES){ + peerMeshGiveUp.add(uuid); + logLine('err', 'peer '+uuid+' mesh failed '+attempts+'x — giving up, audio stays on SFU'); + return; + } + /* exponential backoff: 2s, 4s, 8s, 16s (capped). Keeps the room + * from melting when a peer's NAT genuinely can't mesh through. */ + const delayMs = Math.min(16000, 2000 * Math.pow(2, attempts - 1)); + logLine('', 'peer '+uuid+' failed — reconnecting in '+(delayMs/1000)+'s (attempt '+attempts+'/'+PEER_MESH_MAX_RETRIES+')'); + const t = setTimeout(() => { + peerMeshTimers.delete(uuid); + if (members.has(uuid) && canSpeak(members.get(uuid).role) && canSpeak(myRole)){ + connectToPeer(uuid, myUUID < uuid); + } + }, delayMs); + peerMeshTimers.set(uuid, t); }; if (weOffer){ const offer = await pc.createOffer(); @@ -4127,8 +4171,8 @@ logLine('', 'ready — pick a handle, type a rendezvous code, enter the space');