From 3be8bd2a7885fbd141f6cc12c2d2073f9a0ac56c Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 10:27:45 +0200 Subject: debug(transport): opt-in WebRTC health tracing for mobile-lock investigation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Client (transport.js): a localStorage ring buffer of connection/ICE/ DataChannel state transitions, visibility changes, request timeouts, and periodic health pings — enabled once via ?trace=1 (persists), read back at any time via #mb-debug without devtools. Off by default, zero behavior change unless enabled. Node (webrtc_server.py): MESHBAY_WEBRTC_TRACE=1 gates ICE-state-change logging and a per-session heartbeat (message count, seconds since last message, ICE/connection state) every 30s. Debugging aid for the "stuck after several minutes of mobile screen lock" report — not a fix. Stays on this branch until confirmed useful/resolved. --- .../src/meshbay_hub/static/transport.js | 156 +++++++++++++++++++++ 1 file changed, 156 insertions(+) (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js') diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js index 7535594..6c038f6 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js @@ -81,6 +81,103 @@ const ADMIN_OP_TYPES = new Set([ 'group_detach', 'invite_create', ]); +// ── Diagnostic trace (opt-in, off by default) ─────────────────────────────── +// Ring buffer of transport health events (connection/ICE/DataChannel state +// transitions, request timeouts, visibility changes, periodic health pings), +// persisted to localStorage so a connection that gets stuck can be inspected +// after the fact — the field case this exists for is a phone with no +// devtools attached. Added while chasing a report of the transport going +// unresponsive after a mobile screen lock of several minutes; kept in the +// tree afterward rather than ripped out, since the next hard-to-reproduce +// connection bug will want the same thing and it costs nothing while off. +// +// Enable once by opening the app with ?trace=1 in the URL — this persists in +// localStorage, so every later visit stays in trace mode until ?trace=0 +// clears it. Read the log back at any time by navigating to #mb-debug (e.g. +// https://meshbay.org/app/#mb-debug), which replaces the page with a plain +// text dump — no devtools required. +const TRACE_KEY = 'mb_trace'; +const TRACE_LOG_KEY = 'mb_trace_log'; +const TRACE_MAX = 500; +// How often to probe the channel with a ping while trace mode is on — purely +// diagnostic (to see when a health check starts failing), not a keepalive: +// must stay opt-in, never run by default. +const TRACE_PING_INTERVAL_MS = 25000; + +(function _initTraceFlag() { + try { + const params = new URLSearchParams(location.search); + if (params.has('trace')) { + if (params.get('trace') === '0') localStorage.removeItem(TRACE_KEY); + else localStorage.setItem(TRACE_KEY, '1'); + } + } catch { /* localStorage unavailable (private mode, etc.) — trace stays off */ } +})(); + +function traceEnabled() { + try { return localStorage.getItem(TRACE_KEY) === '1'; } catch { return false; } +} + +function trace(event, data) { + if (!traceEnabled()) return; + try { + const buf = JSON.parse(localStorage.getItem(TRACE_LOG_KEY) || '[]'); + buf.push({ t: new Date().toISOString(), event, ...data }); + while (buf.length > TRACE_MAX) buf.shift(); + localStorage.setItem(TRACE_LOG_KEY, JSON.stringify(buf)); + } catch { /* storage full or unavailable — tracing is best-effort */ } +} + +window.MeshBayTrace = { + enabled: traceEnabled, + dump() { + try { return JSON.parse(localStorage.getItem(TRACE_LOG_KEY) || '[]'); } catch { return []; } + }, + clear() { try { localStorage.removeItem(TRACE_LOG_KEY); } catch { /* ignore */ } }, +}; + +function _showTraceView() { + { + const renderTraceView = () => { + const log = window.MeshBayTrace.dump(); + const text = JSON.stringify(log, null, 2); + document.body.innerHTML = ''; + document.title = 'MeshBay — Diagnostic'; + const bar = document.createElement('div'); + bar.style.cssText = 'font-family:monospace;padding:8px;'; + const copyBtn = document.createElement('button'); + copyBtn.textContent = 'Copier'; + copyBtn.onclick = () => { navigator.clipboard.writeText(text).catch(() => {}); }; + const clearBtn = document.createElement('button'); + clearBtn.textContent = 'Vider'; + clearBtn.onclick = () => { window.MeshBayTrace.clear(); renderTraceView(); }; + const refreshBtn = document.createElement('button'); + refreshBtn.textContent = 'Rafraîchir'; + refreshBtn.onclick = renderTraceView; + const info = document.createElement('span'); + info.textContent = ` — ${log.length} évènement(s) — trace ${traceEnabled() ? 'active' : 'inactive'}`; + info.style.marginLeft = '8px'; + bar.append(copyBtn, clearBtn, refreshBtn, info); + const pre = document.createElement('pre'); + pre.style.cssText = 'font-family:monospace;font-size:11px;white-space:pre-wrap;' + + 'word-break:break-all;padding:8px;'; + pre.textContent = text; + document.body.append(bar, pre); + }; + renderTraceView(); + } +} + +// Fragment-only URL changes (typing #mb-debug into an already-loaded page, +// or a link to it) do not reload the document, so DOMContentLoaded alone +// would miss them — hashchange is what a same-document navigation fires. +if (location.hash === '#mb-debug') { + document.addEventListener('DOMContentLoaded', _showTraceView); +} +window.addEventListener('hashchange', () => { + if (location.hash === '#mb-debug') _showTraceView(); +}); + const JOIN_REFUSALS = { code_required: 'This node does not know this browser yet. Ask the node operator ' + 'for a pairing code (meshbay-node operator pair).', @@ -168,6 +265,7 @@ class MeshBayTransport { this._channel.onopen = () => { clearTimeout(timeout); this._connected = true; + trace('channel_open', {}); resolve(); }; }); @@ -175,6 +273,11 @@ class MeshBayTransport { this._channel.onmessage = (event) => this._onMessage(event.data); this._channel.onclose = (ev) => { console.warn('[MeshBay] DataChannel closed', this._channel?.readyState, ev); + trace('channel_close', { + readyState: this._channel?.readyState, + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + }); this._connected = false; if (channelReject) channelReject(new Error('DataChannel closed')); for (const [, p] of this._pending) p.reject(new Error('DataChannel closed')); @@ -182,16 +285,63 @@ class MeshBayTransport { }; this._channel.onerror = (ev) => { console.error('[MeshBay] DataChannel error', ev); + trace('channel_error', { + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + }); if (channelReject) channelReject(new Error('DataChannel error')); }; this._pc.onconnectionstatechange = () => { console.log('[MeshBay] PC state:', this._pc.connectionState); + trace('pc_state', { state: this._pc.connectionState }); }; this._pc.oniceconnectionstatechange = () => { console.log('[MeshBay] ICE state:', this._pc.iceConnectionState); + trace('ice_state', { state: this._pc.iceConnectionState }); }; + // Diagnostic-only: a periodic health ping and a resume-triggered one, so + // a trace captures exactly what state the connection was in right as the + // page comes back from being backgrounded/locked — never active unless + // trace mode is on (see TRACE_KEY above). + if (traceEnabled()) { + const healthPing = async (reason) => { + const before = { + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + channel: this._channel?.readyState, + }; + const start = Date.now(); + try { + await this.ping(8000); + trace('health_ping', { reason, ok: true, rtt_ms: Date.now() - start, ...before }); + } catch (e) { + trace('health_ping', { reason, ok: false, error: String(e && e.message || e), + elapsed_ms: Date.now() - start, ...before }); + } + }; + const onVisibility = () => { + trace('visibility', { + state: document.visibilityState, + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + channel: this._channel?.readyState, + }); + if (document.visibilityState === 'visible' && this._channel?.readyState === 'open') { + healthPing('resume'); + } + }; + document.addEventListener('visibilitychange', onVisibility); + const healthInterval = setInterval(() => { + if (this._channel?.readyState === 'open') healthPing('interval'); + }, TRACE_PING_INTERVAL_MS); + this._diagCleanup = () => { + document.removeEventListener('visibilitychange', onVisibility); + clearInterval(healthInterval); + }; + } + const offer = await this._pc.createOffer(); await this._pc.setLocalDescription(offer); @@ -1440,6 +1590,7 @@ class MeshBayTransport { get gekRaw() { return this._gekRaw; } close() { + if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; } if (this._channel) this._channel.close(); if (this._pc) this._pc.close(); this._connected = false; @@ -1456,6 +1607,11 @@ class MeshBayTransport { this._pending.delete(id); console.error('[MeshBay] Response timeout for', obj.type, 'after', timeoutMs, 'ms, channel=', this._channel?.readyState); + trace('send_timeout', { + reqType: obj.type, timeoutMs, + pc: this._pc?.connectionState, ice: this._pc?.iceConnectionState, + channel: this._channel?.readyState, + }); reject(new Error('Response timeout')); }, timeoutMs); this._pending.set(id, { -- cgit v1.2.3 From 27d59cbb2d11d863273e2282257d81ac9911ab97 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 11:17:37 +0200 Subject: fix(transport): auto-reconnect after WebRTC failure, without dropping in-flight streams/downloads MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Confirmed live (client trace + node logs, mobile screen-lock ~5min): ICE goes disconnected -> failed within ~10s on both ends, but the DataChannel's readyState stays "open" throughout, so nothing failed fast — every request just sat out its own 8s/30s timeout, matching the reported symptom (poster spinners, blocked chat, dead new streams). transport.js: on connectionState "failed", reject pending requests immediately (TransportLostError) and start a self-contained reconnect loop (capped exponential backoff, redoes the full signaling handshake — the node already discards the old session on its own "failed"/"closed", so there is nothing lower-level to resume). New hooks: onNeedToken (fetch a fresh JWT, since the captured one may have expired during the outage) and onReconnected (let a consumer resume something that was mid-flight). file-utils.js: pipelinedDownload retries a lost chunk instead of aborting the whole transfer — covers Files downloads, poster/thumbnail fetches, and music-player.js's blob-based track download, all of which go through it. video-player.js: onReconnected reissues the existing seek-to-current-time path, which already knows how to land a new stream_init on the live SourceBuffer without resetting playback. Playing audio is unaffected either way — musicbay.md's design downloads a track to a blob before playing it, so a dead transport was never a network dependency for what is already playing. Stays on this branch until confirmed by real-device testing. --- .../src/meshbay_hub/static/file-utils.js | 37 ++++- .../src/meshbay_hub/static/group-page.js | 5 + .../src/meshbay_hub/static/transport.js | 171 ++++++++++++++++++++- .../src/meshbay_hub/static/video-player.js | 19 +++ 4 files changed, 224 insertions(+), 8 deletions(-) (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js') diff --git a/packages/meshbay-hub/src/meshbay_hub/static/file-utils.js b/packages/meshbay-hub/src/meshbay_hub/static/file-utils.js index b44d105..ba76ac9 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/file-utils.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/file-utils.js @@ -117,6 +117,41 @@ function _b64ToU8(b64) { return arr; } +// A dead transport (screen-lock WebRTC failure, see transport.js's +// _reconnectLoop) surfaces here as a rejected fetchChunk — TransportLostError +// when the pending request was killed outright, a plain timeout if it was +// still waiting when this ran. Either way the chunk itself was never the +// problem, and the file already on disk (writable has real bytes in it by +// now) is worth more than an all-or-nothing download: retry the same chunk +// instead of letting one bad moment abort the whole transfer. Each retry +// re-enters transport.fetchChunk, whose own _sendAndWait waits out an +// in-flight reconnect before trying again, so this loop is mostly just +// giving that reconnect the time and the attempts to land. +const CHUNK_RETRY_ATTEMPTS = 6; +const CHUNK_RETRY_DELAY_MS = 1500; + +function _isRetryableTransportError(err) { + return err.name === 'TransportLostError' + || err.message === 'Response timeout' + || (err.message || '').startsWith('DataChannel not open'); +} + +async function _fetchChunkResilient(transport, fileId, index) { + let lastErr; + for (let attempt = 0; attempt < CHUNK_RETRY_ATTEMPTS; attempt++) { + try { + return await transport.fetchChunk(fileId, index); + } catch (err) { + if (!_isRetryableTransportError(err)) throw err; + lastErr = err; + if (attempt < CHUNK_RETRY_ATTEMPTS - 1) { + await new Promise((r) => setTimeout(r, CHUNK_RETRY_DELAY_MS)); + } + } + } + throw lastErr; +} + async function pipelinedDownload(transport, gekKey, fileId, totalChunks, onChunk, writable, signal) { const results = writable ? null : new Array(totalChunks); @@ -125,7 +160,7 @@ async function pipelinedDownload(transport, gekKey, fileId, totalChunks, onChunk const fire = () => { while (nextSend < totalChunks && nextSend - nextRecv < PIPELINE_WINDOW) { - inflight[nextSend] = transport.fetchChunk(fileId, nextSend); + inflight[nextSend] = _fetchChunkResilient(transport, fileId, nextSend); nextSend++; } }; diff --git a/packages/meshbay-hub/src/meshbay_hub/static/group-page.js b/packages/meshbay-hub/src/meshbay_hub/static/group-page.js index cdbe612..8683a70 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/group-page.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/group-page.js @@ -226,6 +226,11 @@ function GroupPage({ groupId, group, token, username, userId, userPrefs, // any other, and two sources for one address is how they drift. const transport = new window.MeshBayTransport(HUB, live); transportRef.current = transport; + // Consulted only by the automatic reconnect after a WebRTC failure + // (transport.js's _reconnectLoop) — the token captured by this + // connect() call can be stale by then, since the whole point is that + // some real time (screen lock, a dead NAT mapping) passed unnoticed. + transport.onNeedToken = async () => (await ensureFreshToken()) || token; const ack = await transport.connect( nodeId, live, groupId, null, sessionKeys, session.bundleKey, username, diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js index 6c038f6..56ec2c7 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js @@ -214,6 +214,29 @@ class MeshBayTransport { // several uploads may be in flight at once and their acks interleave; the // node names the file in every one. this._uploaders = new Map(); + // Set once close() runs — stops the automatic reconnect from firing on a + // connection the caller tore down on purpose (leaving the group, page + // unload), which would otherwise race back in right as everything else + // is being torn down. + this._closed = false; + // The arguments connect() was last given, minus the token (refreshed at + // reconnect time — see onNeedToken) and sessionKeys (kept live on `this`, + // since a reconnect must reuse the identity connect() settled on, not + // whatever the very first caller passed in — see _reconnectLoop). + this._connectArgs = null; + this._lastToken = null; + this._reconnectPromise = null; + this._reconnectAttempts = 0; + // True only for the duration of the connect() call _reconnectLoop makes + // to actually retry — as opposed to the backoff delay around it, which + // is most of _reconnectPromise's lifetime. Needed because that connect() + // call sends its own handshake through _sendAndWait, which would + // otherwise see the very _reconnectPromise it is running inside of as + // "a reconnect to wait for" and stall every handshake step for the full + // 6s gate below before ever sending it. + this._inReconnectAttempt = false; + this._onReconnected = null; + this._onNeedToken = null; } get connected() { return this._connected; } @@ -235,6 +258,17 @@ class MeshBayTransport { set onMusicbrainzConfig(fn) { this._onMusicbrainzConfig = fn; } set onMusicbrainzEnabled(fn) { this._onMusicbrainzEnabled = fn; } set onIndexProgress(fn) { this._onIndexProgress = fn; } + // Fired once an automatic reconnect (see _reconnectLoop) lands a fresh + // handshake, so a consumer with something mid-flight on the old channel — + // today only the video player — can pick back up rather than sit dead. + set onReconnected(fn) { this._onReconnected = fn; } + // Reconnecting redoes the handshake, which needs a JWT that may have gone + // stale while the connection was down for minutes. Without this the + // reconnect resends whatever token the original connect() call captured, + // which the node's clock-skew check (stale_request) or plain expiry can + // by then have already invalidated. Set to whatever the caller uses to + // refresh the hub session token (see group-page.js's ensureFreshToken). + set onNeedToken(fn) { this._onNeedToken = fn; } get sessionKeys() { return this._sessionKeys; } @@ -244,6 +278,12 @@ class MeshBayTransport { async connect(nodeId, jwtToken, groupId, gekRaw, sessionKeys, bundleKey, username, userId, joinCode) { + // Remembered for _reconnectLoop, which calls connect() again with these + // same values (plus a freshly-fetched token and the identity connect() + // itself settles on below) after the WebRTC connection is declared + // "failed" — see the pc.onconnectionstatechange handler further down. + this._connectArgs = { nodeId, groupId, gekRaw, bundleKey, username, userId, joinCode }; + this._lastToken = jwtToken; this._gekRaw = gekRaw || null; this._sessionKeys = sessionKeys || null; this._bundleKey = bundleKey || null; @@ -292,13 +332,35 @@ class MeshBayTransport { if (channelReject) channelReject(new Error('DataChannel error')); }; - this._pc.onconnectionstatechange = () => { - console.log('[MeshBay] PC state:', this._pc.connectionState); - trace('pc_state', { state: this._pc.connectionState }); + // Captured locally rather than read back through `this._pc`: once a + // reconnect replaces it, a late event from this (by then orphaned) pc + // must still be judged against the pc it actually came from, not + // whatever is current — the `pc === this._pc` check below is what that + // buys. + const pc = this._pc; + pc.onconnectionstatechange = () => { + console.log('[MeshBay] PC state:', pc.connectionState); + trace('pc_state', { state: pc.connectionState }); + // "failed" is ICE's own verdict that nothing here will recover on its + // own (unlike a transient "disconnected", which often clears itself) — + // confirmed live: mobile screen lock for several minutes reliably + // produces disconnected → failed about 10s apart, on both ends, and + // nothing today ever moves past that without a full page reload. + // `channel.readyState` is no help distinguishing this: it was observed + // staying "open" throughout, so every send from here on would simply + // sit out its own timeout instead of failing fast. + if (pc.connectionState === 'failed' && pc === this._pc && !this._closed) { + this._connected = false; + this._reconnect(); + const err = new Error('WebRTC connection lost'); + err.name = 'TransportLostError'; + for (const [, p] of this._pending) p.reject(err); + this._pending.clear(); + } }; - this._pc.oniceconnectionstatechange = () => { - console.log('[MeshBay] ICE state:', this._pc.iceConnectionState); - trace('ice_state', { state: this._pc.iceConnectionState }); + pc.oniceconnectionstatechange = () => { + console.log('[MeshBay] ICE state:', pc.iceConnectionState); + trace('ice_state', { state: pc.iceConnectionState }); }; // Diagnostic-only: a periodic health ping and a resume-triggered one, so @@ -575,6 +637,82 @@ class MeshBayTransport { throw rejected; } + /** + * Kick off (or join, if one is already running) the automatic reconnect + * after the WebRTC connection is declared unrecoverable. Idempotent: every + * caller racing to reconnect at once — the connectionstatechange handler, + * and any request that lands in the gap _sendAndWait waits out below — + * shares the one attempt instead of piling up parallel handshakes against + * the node. + */ + _reconnect() { + if (this._closed) return Promise.resolve(); + if (!this._reconnectPromise) { + this._reconnectPromise = this._reconnectLoop().finally(() => { + this._reconnectPromise = null; + }); + } + return this._reconnectPromise; + } + + /** + * Redo the signaling handshake from scratch — the only thing that works + * once aiortc has declared a connection "failed": the node discards that + * session the moment it sees the same state (webrtc_server.py's + * on_state_change), so there is no lower-level session left to resume, only + * a fresh one to negotiate. Retries with capped exponential backoff + * (1s, 2s, 4s ... 30s) rather than a fixed number of attempts, because the + * two real causes seen so far — a mobile carrier dropping the NAT mapping + * during screen lock, and the node's own machine being briefly unreachable + * — both resolve on their own eventually, and there is no good moment to + * decide the user would rather see a dead app than keep waiting. + */ + async _reconnectLoop() { + this._reconnectAttempts = 0; + while (!this._closed) { + this._reconnectAttempts += 1; + const delayMs = Math.min(30000, 1000 * 2 ** (this._reconnectAttempts - 1)); + trace('reconnect_wait', { attempt: this._reconnectAttempts, delay_ms: delayMs }); + await new Promise((r) => setTimeout(r, delayMs)); + if (this._closed) return; + try { + // Best-effort: these are already unusable, but leaving them wired up + // risks a stray late event from the old pc doing something once a + // new one is in `this._pc` — the `pc === this._pc` guard above closes + // most of that gap, this closes the rest. + try { this._channel && this._channel.close(); } catch { /* already gone */ } + try { this._pc && this._pc.close(); } catch { /* already gone */ } + const args = this._connectArgs; + const token = this._onNeedToken ? await this._onNeedToken() : this._lastToken; + trace('reconnect_attempt', { attempt: this._reconnectAttempts }); + this._inReconnectAttempt = true; + try { + await this.connect(args.nodeId, token, args.groupId, args.gekRaw, + this._sessionKeys, args.bundleKey, args.username, + args.userId, args.joinCode); + } finally { + this._inReconnectAttempt = false; + } + trace('reconnect_ok', { attempt: this._reconnectAttempts }); + console.log('[MeshBay] Reconnected after', this._reconnectAttempts, 'attempt(s)'); + if (this._onReconnected) { + try { this._onReconnected(); } catch (e) { + console.error('[MeshBay] onReconnected handler threw:', e); + } + } + return; + } catch (e) { + trace('reconnect_attempt_failed', { + attempt: this._reconnectAttempts, error: String(e && e.message || e), + }); + console.warn('[MeshBay] Reconnect attempt', this._reconnectAttempts, + 'failed:', e.message); + // Loop again with a longer backoff — closing over `args`/`token` + // freshly next time, in case the token was the actual problem. + } + } + } + /** * Pair this browser with the node using a one-time code (M3, and the same * substitution as H3). @@ -1590,6 +1728,11 @@ class MeshBayTransport { get gekRaw() { return this._gekRaw; } close() { + // Must be set before pc.close() below: that close() itself can drive the + // pc to "closed" synchronously, and the connectionstatechange handler + // only skips reconnecting because of this flag, not because "closed" is + // absent from its own trigger condition. + this._closed = true; if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; } if (this._channel) this._channel.close(); if (this._pc) this._pc.close(); @@ -1600,7 +1743,21 @@ class MeshBayTransport { // ── Internal ────────────────────────────────────────────────────────────── - _sendAndWait(obj, timeoutMs = 30000) { + async _sendAndWait(obj, timeoutMs = 30000) { + // A reconnect already in flight (see _reconnectLoop) means the channel + // this would send on is the one just declared dead. Waiting here, bounded + // rather than open-ended — the loop can be backing off for up to 30s + // between attempts, and a caller (in particular file-utils.js's + // pipelinedDownload, which retries a lost chunk itself) should get its + // own timeout rather than sit through someone else's backoff — gives a + // fresh handshake a real chance to land before this request is even + // attempted, instead of guaranteeing it dies with the old one. + if (this._reconnectPromise && !this._inReconnectAttempt) { + await Promise.race([ + this._reconnectPromise.catch(() => {}), + new Promise((r) => setTimeout(r, 6000)), + ]); + } return new Promise((resolve, reject) => { const id = this._seqId++; const timeout = setTimeout(() => { diff --git a/packages/meshbay-hub/src/meshbay_hub/static/video-player.js b/packages/meshbay-hub/src/meshbay_hub/static/video-player.js index 82e1116..78760d8 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/video-player.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/video-player.js @@ -529,6 +529,24 @@ function VideoPlayer({ entry, transportRef, gekRef, onClose, onDownload }) { setPhase('error'); }; + // The old stream died with the connection (the node retires it the + // moment its session goes away — see webrtc_server.py's + // on_state_change), so there is nothing to resume on the wire, only a + // reason to ask again. requestSeek already knows how to land a new + // stream_init on the live SourceBuffer without resetting playback — + // exactly what dragging the scrubber does — so reusing it here means a + // screen-lock reconnect looks like a seek to where the film already + // was, not a reload. + transport.onReconnected = () => { + if (cancelled) return; + const v = videoRef.current; + const seek = requestSeekRef.current; + if (!v || !seek) return; + console.log('[MeshBay] transport reconnected — resuming stream at', + v.currentTime.toFixed(1)); + seek(v.currentTime); + }; + transport.onStreamInit = (msg) => { if (cancelled) return; if (msg.file_id && msg.file_id !== entry.id) return; @@ -849,6 +867,7 @@ function VideoPlayer({ entry, transportRef, gekRef, onClose, onDownload }) { transport.onStreamData = null; transport.onStreamEnd = null; transport.onStreamError = null; + transport.onReconnected = null; } // The queue can hold several megabytes of decrypted video. queueRef.current = []; -- cgit v1.2.3 From 6e0d2b9a477f1e9f1dec646c0de15440106ed7ed Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 11:44:23 +0200 Subject: fix(music): don't throw "Transport not connected" while a reconnect is landing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found live: >5min screen lock while music played, then "Transport not connected" at the 2nd track's end (~8min in) despite the auto-reconnect from the previous commit. Root cause: fetchTrackBlob's own `!transport.connected` pre-flight check ran and threw before the already-in-progress reconnect got the few seconds it needed — it never reached _sendAndWait, which is the only place the previous fix taught the transport to wait. Adds transport.waitForReconnect(), factored out of _sendAndWait's existing gate, and calls it from fetchTrackBlob before giving up. No-op when nothing is being reconnected, so the ordinary path is unchanged. --- .../src/meshbay_hub/static/music-player.js | 9 ++++- .../src/meshbay_hub/static/transport.js | 43 +++++++++++++++------- 2 files changed, 38 insertions(+), 14 deletions(-) (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js') diff --git a/packages/meshbay-hub/src/meshbay_hub/static/music-player.js b/packages/meshbay-hub/src/meshbay_hub/static/music-player.js index 7a3978c..73cae27 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/music-player.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/music-player.js @@ -218,7 +218,14 @@ function MusicPlayerBar({ transportRef, gekRef, queue, onClose }) { const cached = blobCacheRef.current.get(entry.id); if (cached) return cached.url; const transport = transportRef.current; - if (!transport || !transport.connected) throw new Error(t('music.err_transport')); + if (!transport) throw new Error(t('music.err_transport')); + // A track ending (or "next") right after a screen-lock reconnect started + // is exactly when this used to throw: `connected` was still false because + // the reconnect it only had to wait a few seconds for hadn't landed yet. + // waitForReconnect is a no-op when nothing is in flight, so this costs + // nothing on the ordinary path. + if (!transport.connected) await transport.waitForReconnect(); + if (!transport.connected) throw new Error(t('music.err_transport')); let downloadId = entry.id; let downloadSize = entry.size; diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js index 56ec2c7..4eb7ac2 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js @@ -1727,6 +1727,30 @@ class MeshBayTransport { get gekRaw() { return this._gekRaw; } + /** + * Give an automatic reconnect already in progress (see _reconnectLoop) a + * bounded chance to land before giving up. + * + * _sendAndWait does this internally for every request that goes through + * it, so most callers never need this directly. It exists for the ones + * that check `transport.connected` themselves before doing anything else — + * music-player.js's fetchTrackBlob is the one this was written for: found + * live throwing "Transport not connected" on the track *after* a + * screen-lock reconnect had already been under way for a while, because + * that check ran, saw `connected` still false, and threw before the + * reconnect it only had to wait a few seconds for got the chance to finish. + * A no-op — returns immediately — when nothing is being reconnected, + * including once one has already succeeded, so it is safe to call + * unconditionally ahead of such a check. + */ + async waitForReconnect(timeoutMs = 6000) { + if (!this._reconnectPromise) return; + await Promise.race([ + this._reconnectPromise.catch(() => {}), + new Promise((r) => setTimeout(r, timeoutMs)), + ]); + } + close() { // Must be set before pc.close() below: that close() itself can drive the // pc to "closed" synchronously, and the connectionstatechange handler @@ -1745,19 +1769,12 @@ class MeshBayTransport { async _sendAndWait(obj, timeoutMs = 30000) { // A reconnect already in flight (see _reconnectLoop) means the channel - // this would send on is the one just declared dead. Waiting here, bounded - // rather than open-ended — the loop can be backing off for up to 30s - // between attempts, and a caller (in particular file-utils.js's - // pipelinedDownload, which retries a lost chunk itself) should get its - // own timeout rather than sit through someone else's backoff — gives a - // fresh handshake a real chance to land before this request is even - // attempted, instead of guaranteeing it dies with the old one. - if (this._reconnectPromise && !this._inReconnectAttempt) { - await Promise.race([ - this._reconnectPromise.catch(() => {}), - new Promise((r) => setTimeout(r, 6000)), - ]); - } + // this would send on is the one just declared dead. `_inReconnectAttempt` + // excludes the handshake connect() itself makes while reconnecting — that + // call runs *inside* this same _reconnectPromise, which cannot resolve + // until it returns, so waiting on it here would just be waiting on + // itself for the full 6s, on every step of the handshake, every time. + if (!this._inReconnectAttempt) await this.waitForReconnect(6000); return new Promise((resolve, reject) => { const id = this._seqId++; const timeout = setTimeout(() => { -- cgit v1.2.3 From 5dcf3066a2be83d5422ea176e39b413c4769bcf8 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 12:44:50 +0200 Subject: fix(transport): wake a backing-off reconnect on visibilitychange, fix listener leak MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Confirmed live by trace: during a screen lock, every reconnect attempt failed with "Failed to fetch" (the browser grants no network access to a locked/backgrounded tab, no code can change that) — expected. But the backoff timer itself was also throttled while locked: an attempt scheduled 30s out took ~3 minutes of wall clock to fire, because a backgrounded tab's timers run only when the OS lets them. Recovery after unlocking was correspondingly delayed rather than prompt. Fix: an always-on visibilitychange listener (separate from the diagnostic one, and unlike it not gated on trace mode) resolves the current backoff wait immediately once the page is visible again, instead of waiting out whatever of it is left. The actual reconnect this enables is fast (~1.2s in the trace that showed the "Failed to fetch" run) — the wait was the throttled part. Also fixes a real bug the same trace exposed: connect() re-arms the diagnostic visibility listener and health-ping interval on every attempt without ever removing the previous instance's — 8 failed attempts during one lock left 8 duplicate `visibility` trace lines per real event, and (more than a cosmetic issue) 8 concurrent health-ping intervals once reconnected. --- .../src/meshbay_hub/static/transport.js | 40 +++++++++++++++++++++- 1 file changed, 39 insertions(+), 1 deletion(-) (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js') diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js index 4eb7ac2..35f729e 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js @@ -237,6 +237,27 @@ class MeshBayTransport { this._inReconnectAttempt = false; this._onReconnected = null; this._onNeedToken = null; + // Cuts the backoff wait short the moment the page is foregrounded again — + // found live to matter: a screen lock throttles the tab's own timers + // along with everything else, so a backoff already counting down when the + // phone locked can run for minutes of *wall clock* past its nominal delay + // before it next gets to run at all. Set once, here, rather than inside + // connect() like the diagnostic listener above it — this one has to + // survive every reconnect attempt, not restart with each one. + this._reconnectWakeResolve = null; + this._onVisibilityWake = () => { + if (document.visibilityState === 'visible') this._wakeReconnect(); + }; + document.addEventListener('visibilitychange', this._onVisibilityWake); + } + + /** Cuts short a reconnect currently backing off (see _reconnectLoop). A + * no-op when nothing is waiting, so this is safe to call unconditionally. */ + _wakeReconnect() { + if (this._reconnectWakeResolve) { + this._reconnectWakeResolve(); + this._reconnectWakeResolve = null; + } } get connected() { return this._connected; } @@ -367,6 +388,14 @@ class MeshBayTransport { // a trace captures exactly what state the connection was in right as the // page comes back from being backgrounded/locked — never active unless // trace mode is on (see TRACE_KEY above). + // + // connect() runs again on every reconnect attempt (see _reconnectLoop), + // and each run used to add its own listener/interval on top of the + // previous one without ever removing it — confirmed live: 8 failed + // attempts during one screen lock left 8 duplicate `visibility` trace + // lines firing off the same real event. Disposing of the prior instance + // first is what keeps this to one. + if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; } if (traceEnabled()) { const healthPing = async (reason) => { const before = { @@ -673,7 +702,14 @@ class MeshBayTransport { this._reconnectAttempts += 1; const delayMs = Math.min(30000, 1000 * 2 ** (this._reconnectAttempts - 1)); trace('reconnect_wait', { attempt: this._reconnectAttempts, delay_ms: delayMs }); - await new Promise((r) => setTimeout(r, delayMs)); + // Interruptible: _wakeReconnect (fired on visibilitychange → visible) + // resolves this immediately instead of waiting out the rest of a + // backoff that was mostly spent while nothing could succeed anyway. + await new Promise((resolve) => { + const timer = setTimeout(resolve, delayMs); + this._reconnectWakeResolve = () => { clearTimeout(timer); resolve(); }; + }); + this._reconnectWakeResolve = null; if (this._closed) return; try { // Best-effort: these are already unusable, but leaving them wired up @@ -1757,6 +1793,8 @@ class MeshBayTransport { // only skips reconnecting because of this flag, not because "closed" is // absent from its own trigger condition. this._closed = true; + document.removeEventListener('visibilitychange', this._onVisibilityWake); + this._wakeReconnect(); if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; } if (this._channel) this._channel.close(); if (this._pc) this._pc.close(); -- cgit v1.2.3 From a31b26df45860af16061fb43ccc3443381ed3df2 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 13:54:56 +0200 Subject: fix(transport): reconnect used a fresh handshake token but a stale signaling one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found live: every reconnect attempt failed "Signaling failed: 401 Invalid or expired token", looping for 4+ minutes with no chance of ever succeeding. connect() takes one token but uses it in two places — the handshake sent to the node, and the Authorization header on the signaling POST to the hub — and only the constructor's original `this._accessToken` was ever used for the latter. onNeedToken correctly fetches a fresh token for each reconnect attempt, but it only ever reached the handshake; the signaling call kept sending whatever token the transport was constructed with, no matter how many minutes had passed or how many attempts fetched a new one. connect() now updates this._accessToken on every call, reconnects included, so both places use the same current token. --- packages/meshbay-hub/src/meshbay_hub/static/transport.js | 9 +++++++++ 1 file changed, 9 insertions(+) (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js') diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js index 35f729e..9f85ee1 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js @@ -305,6 +305,15 @@ class MeshBayTransport { // "failed" — see the pc.onconnectionstatechange handler further down. this._connectArgs = { nodeId, groupId, gekRaw, bundleKey, username, userId, joinCode }; this._lastToken = jwtToken; + // The constructor sets this once from whatever token the caller had at + // the time — and the signaling POST below reads *this*, not `jwtToken`. + // A reconnect passes a freshly-fetched `jwtToken` (see onNeedToken) but + // that never reached here before, so the signaling call kept using the + // original token no matter how many minutes had passed or how many + // reconnect attempts fetched a new one — confirmed live: every attempt + // failed "Signaling failed: 401 Invalid or expired token" in a loop, + // never actually trying the fresh token connect() had just been handed. + this._accessToken = jwtToken; this._gekRaw = gekRaw || null; this._sessionKeys = sessionKeys || null; this._bundleKey = bundleKey || null; -- cgit v1.2.3