diff options
| author | Christophe Besson <cbesson@gmail.com> | 2026-08-26 15:09:20 +0200 |
|---|---|---|
| committer | Christophe Besson <cbesson@gmail.com> | 2026-08-26 15:09:20 +0200 |
| commit | db7f81fd08742847f3ebea061c75530b7b31b934 (patch) | |
| tree | 14b308686860eea0a9c12c47c39f8a68a1bba376 /packages/meshbay-hub/src/meshbay_hub/static/transport.js | |
| parent | 7126fd3c265ba75d77b449bfd0f83f5f3e584b74 (diff) | |
| parent | 59d9f50bf41b2b38b7c95f8da52b698dff923c2d (diff) | |
| download | meshbay-db7f81fd08742847f3ebea061c75530b7b31b934.tar.gz | |
Merge branch 'debug/webrtc-lock-resume'
WebRTC transport dies silently after an extended mobile screen lock
(confirmed live via client-side trace + node logs): ICE goes
disconnected -> failed within ~10s of each other on both ends, but the
DataChannel's readyState stays "open" throughout, so nothing failed fast —
every request just sat out its own timeout, matching the reported symptom
(poster spinners, blocked chat, dead new streams, stuck music).
- Automatic reconnect on WebRTC "failed": capped exponential backoff,
redoes the full signaling handshake, wakes immediately on
visibilitychange instead of waiting out a throttled backoff timer.
- Fixed two real bugs the reconnect work exposed: the signaling POST to
the hub kept using the token captured at construction, never the fresh
one fetched per reconnect attempt (401 loop, no possible recovery); and
connect() re-armed a diagnostic listener/interval on every attempt
without disposing of the previous one.
- pipelinedDownload retries a lost chunk instead of aborting the whole
transfer — covers Files downloads, video poster/thumbnail fetches, and
music-player.js's blob-based track download.
- music-player.js: don't throw "Transport not connected" while a
reconnect is already landing (waitForReconnect); prefetch depth now
adapts to network type (5 tracks ahead on Wi-Fi, 3 on cellular or
unrecognized — Firefox/Safari included, where the detection API is
simply absent).
- video-player.js: onReconnected reissues the existing seek-to-current-time
path, so a mid-stream reconnect looks like an ordinary seek rather than
a dead player; holds a Screen Wake Lock unconditionally while open.
- New opt-in (off by default) user preference: keep the screen on during
audio playback, for whoever wants to trade battery for sidestepping the
screen-lock gap entirely — off by default because the ordinary
expectation (matching Spotify/Deezer) is that the phone locks on its
own while listening.
- hub: /app and / now serve Cache-Control: no-store — the SPA shell had no
cache header at all, so a browser that cached it heuristically could
keep re-serving an old build (old ASSET_V, old JS) through any number of
reloads or pull-to-refreshes.
Verified against real production use across many rounds (demo groups,
actual mobile screen-lock testing) rather than synthetic reproduction
alone. 430 hub tests + 650 node tests passing throughout.
Diffstat (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js')
| -rw-r--r-- | packages/meshbay-hub/src/meshbay_hub/static/transport.js | 387 |
1 files changed, 382 insertions, 5 deletions
diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js index 7535594..9f85ee1 100644 --- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js +++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js @@ -81,6 +81,103 @@ const ADMIN_OP_TYPES = new Set([ 'group_detach', 'invite_create', ]); +// ── Diagnostic trace (opt-in, off by default) ─────────────────────────────── +// Ring buffer of transport health events (connection/ICE/DataChannel state +// transitions, request timeouts, visibility changes, periodic health pings), +// persisted to localStorage so a connection that gets stuck can be inspected +// after the fact — the field case this exists for is a phone with no +// devtools attached. Added while chasing a report of the transport going +// unresponsive after a mobile screen lock of several minutes; kept in the +// tree afterward rather than ripped out, since the next hard-to-reproduce +// connection bug will want the same thing and it costs nothing while off. +// +// Enable once by opening the app with ?trace=1 in the URL — this persists in +// localStorage, so every later visit stays in trace mode until ?trace=0 +// clears it. Read the log back at any time by navigating to #mb-debug (e.g. +// https://meshbay.org/app/#mb-debug), which replaces the page with a plain +// text dump — no devtools required. +const TRACE_KEY = 'mb_trace'; +const TRACE_LOG_KEY = 'mb_trace_log'; +const TRACE_MAX = 500; +// How often to probe the channel with a ping while trace mode is on — purely +// diagnostic (to see when a health check starts failing), not a keepalive: +// must stay opt-in, never run by default. +const TRACE_PING_INTERVAL_MS = 25000; + +(function _initTraceFlag() { + try { + const params = new URLSearchParams(location.search); + if (params.has('trace')) { + if (params.get('trace') === '0') localStorage.removeItem(TRACE_KEY); + else localStorage.setItem(TRACE_KEY, '1'); + } + } catch { /* localStorage unavailable (private mode, etc.) — trace stays off */ } +})(); + +function traceEnabled() { + try { return localStorage.getItem(TRACE_KEY) === '1'; } catch { return false; } +} + +function trace(event, data) { + if (!traceEnabled()) return; + try { + const buf = JSON.parse(localStorage.getItem(TRACE_LOG_KEY) || '[]'); + buf.push({ t: new Date().toISOString(), event, ...data }); + while (buf.length > TRACE_MAX) buf.shift(); + localStorage.setItem(TRACE_LOG_KEY, JSON.stringify(buf)); + } catch { /* storage full or unavailable — tracing is best-effort */ } +} + +window.MeshBayTrace = { + enabled: traceEnabled, + dump() { + try { return JSON.parse(localStorage.getItem(TRACE_LOG_KEY) || '[]'); } catch { return []; } + }, + clear() { try { localStorage.removeItem(TRACE_LOG_KEY); } catch { /* ignore */ } }, +}; + +function _showTraceView() { + { + const renderTraceView = () => { + const log = window.MeshBayTrace.dump(); + const text = JSON.stringify(log, null, 2); + document.body.innerHTML = ''; + document.title = 'MeshBay — Diagnostic'; + const bar = document.createElement('div'); + bar.style.cssText = 'font-family:monospace;padding:8px;'; + const copyBtn = document.createElement('button'); + copyBtn.textContent = 'Copier'; + copyBtn.onclick = () => { navigator.clipboard.writeText(text).catch(() => {}); }; + const clearBtn = document.createElement('button'); + clearBtn.textContent = 'Vider'; + clearBtn.onclick = () => { window.MeshBayTrace.clear(); renderTraceView(); }; + const refreshBtn = document.createElement('button'); + refreshBtn.textContent = 'Rafraîchir'; + refreshBtn.onclick = renderTraceView; + const info = document.createElement('span'); + info.textContent = ` — ${log.length} évènement(s) — trace ${traceEnabled() ? 'active' : 'inactive'}`; + info.style.marginLeft = '8px'; + bar.append(copyBtn, clearBtn, refreshBtn, info); + const pre = document.createElement('pre'); + pre.style.cssText = 'font-family:monospace;font-size:11px;white-space:pre-wrap;' + + 'word-break:break-all;padding:8px;'; + pre.textContent = text; + document.body.append(bar, pre); + }; + renderTraceView(); + } +} + +// Fragment-only URL changes (typing #mb-debug into an already-loaded page, +// or a link to it) do not reload the document, so DOMContentLoaded alone +// would miss them — hashchange is what a same-document navigation fires. +if (location.hash === '#mb-debug') { + document.addEventListener('DOMContentLoaded', _showTraceView); +} +window.addEventListener('hashchange', () => { + if (location.hash === '#mb-debug') _showTraceView(); +}); + const JOIN_REFUSALS = { code_required: 'This node does not know this browser yet. Ask the node operator ' + 'for a pairing code (meshbay-node operator pair).', @@ -117,6 +214,50 @@ class MeshBayTransport { // several uploads may be in flight at once and their acks interleave; the // node names the file in every one. this._uploaders = new Map(); + // Set once close() runs — stops the automatic reconnect from firing on a + // connection the caller tore down on purpose (leaving the group, page + // unload), which would otherwise race back in right as everything else + // is being torn down. + this._closed = false; + // The arguments connect() was last given, minus the token (refreshed at + // reconnect time — see onNeedToken) and sessionKeys (kept live on `this`, + // since a reconnect must reuse the identity connect() settled on, not + // whatever the very first caller passed in — see _reconnectLoop). + this._connectArgs = null; + this._lastToken = null; + this._reconnectPromise = null; + this._reconnectAttempts = 0; + // True only for the duration of the connect() call _reconnectLoop makes + // to actually retry — as opposed to the backoff delay around it, which + // is most of _reconnectPromise's lifetime. Needed because that connect() + // call sends its own handshake through _sendAndWait, which would + // otherwise see the very _reconnectPromise it is running inside of as + // "a reconnect to wait for" and stall every handshake step for the full + // 6s gate below before ever sending it. + this._inReconnectAttempt = false; + this._onReconnected = null; + this._onNeedToken = null; + // Cuts the backoff wait short the moment the page is foregrounded again — + // found live to matter: a screen lock throttles the tab's own timers + // along with everything else, so a backoff already counting down when the + // phone locked can run for minutes of *wall clock* past its nominal delay + // before it next gets to run at all. Set once, here, rather than inside + // connect() like the diagnostic listener above it — this one has to + // survive every reconnect attempt, not restart with each one. + this._reconnectWakeResolve = null; + this._onVisibilityWake = () => { + if (document.visibilityState === 'visible') this._wakeReconnect(); + }; + document.addEventListener('visibilitychange', this._onVisibilityWake); + } + + /** Cuts short a reconnect currently backing off (see _reconnectLoop). A + * no-op when nothing is waiting, so this is safe to call unconditionally. */ + _wakeReconnect() { + if (this._reconnectWakeResolve) { + this._reconnectWakeResolve(); + this._reconnectWakeResolve = null; + } } get connected() { return this._connected; } @@ -138,6 +279,17 @@ class MeshBayTransport { set onMusicbrainzConfig(fn) { this._onMusicbrainzConfig = fn; } set onMusicbrainzEnabled(fn) { this._onMusicbrainzEnabled = fn; } set onIndexProgress(fn) { this._onIndexProgress = fn; } + // Fired once an automatic reconnect (see _reconnectLoop) lands a fresh + // handshake, so a consumer with something mid-flight on the old channel — + // today only the video player — can pick back up rather than sit dead. + set onReconnected(fn) { this._onReconnected = fn; } + // Reconnecting redoes the handshake, which needs a JWT that may have gone + // stale while the connection was down for minutes. Without this the + // reconnect resends whatever token the original connect() call captured, + // which the node's clock-skew check (stale_request) or plain expiry can + // by then have already invalidated. Set to whatever the caller uses to + // refresh the hub session token (see group-page.js's ensureFreshToken). + set onNeedToken(fn) { this._onNeedToken = fn; } get sessionKeys() { return this._sessionKeys; } @@ -147,6 +299,21 @@ class MeshBayTransport { async connect(nodeId, jwtToken, groupId, gekRaw, sessionKeys, bundleKey, username, userId, joinCode) { + // Remembered for _reconnectLoop, which calls connect() again with these + // same values (plus a freshly-fetched token and the identity connect() + // itself settles on below) after the WebRTC connection is declared + // "failed" — see the pc.onconnectionstatechange handler further down. + this._connectArgs = { nodeId, groupId, gekRaw, bundleKey, username, userId, joinCode }; + this._lastToken = jwtToken; + // The constructor sets this once from whatever token the caller had at + // the time — and the signaling POST below reads *this*, not `jwtToken`. + // A reconnect passes a freshly-fetched `jwtToken` (see onNeedToken) but + // that never reached here before, so the signaling call kept using the + // original token no matter how many minutes had passed or how many + // reconnect attempts fetched a new one — confirmed live: every attempt + // failed "Signaling failed: 401 Invalid or expired token" in a loop, + // never actually trying the fresh token connect() had just been handed. + this._accessToken = jwtToken; this._gekRaw = gekRaw || null; this._sessionKeys = sessionKeys || null; this._bundleKey = bundleKey || null; @@ -168,6 +335,7 @@ class MeshBayTransport { this._channel.onopen = () => { clearTimeout(timeout); this._connected = true; + trace('channel_open', {}); resolve(); }; }); @@ -175,6 +343,11 @@ class MeshBayTransport { this._channel.onmessage = (event) => this._onMessage(event.data); this._channel.onclose = (ev) => { console.warn('[MeshBay] DataChannel closed', this._channel?.readyState, ev); + trace('channel_close', { + readyState: this._channel?.readyState, + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + }); this._connected = false; if (channelReject) channelReject(new Error('DataChannel closed')); for (const [, p] of this._pending) p.reject(new Error('DataChannel closed')); @@ -182,16 +355,93 @@ class MeshBayTransport { }; this._channel.onerror = (ev) => { console.error('[MeshBay] DataChannel error', ev); + trace('channel_error', { + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + }); if (channelReject) channelReject(new Error('DataChannel error')); }; - this._pc.onconnectionstatechange = () => { - console.log('[MeshBay] PC state:', this._pc.connectionState); + // Captured locally rather than read back through `this._pc`: once a + // reconnect replaces it, a late event from this (by then orphaned) pc + // must still be judged against the pc it actually came from, not + // whatever is current — the `pc === this._pc` check below is what that + // buys. + const pc = this._pc; + pc.onconnectionstatechange = () => { + console.log('[MeshBay] PC state:', pc.connectionState); + trace('pc_state', { state: pc.connectionState }); + // "failed" is ICE's own verdict that nothing here will recover on its + // own (unlike a transient "disconnected", which often clears itself) — + // confirmed live: mobile screen lock for several minutes reliably + // produces disconnected → failed about 10s apart, on both ends, and + // nothing today ever moves past that without a full page reload. + // `channel.readyState` is no help distinguishing this: it was observed + // staying "open" throughout, so every send from here on would simply + // sit out its own timeout instead of failing fast. + if (pc.connectionState === 'failed' && pc === this._pc && !this._closed) { + this._connected = false; + this._reconnect(); + const err = new Error('WebRTC connection lost'); + err.name = 'TransportLostError'; + for (const [, p] of this._pending) p.reject(err); + this._pending.clear(); + } }; - this._pc.oniceconnectionstatechange = () => { - console.log('[MeshBay] ICE state:', this._pc.iceConnectionState); + pc.oniceconnectionstatechange = () => { + console.log('[MeshBay] ICE state:', pc.iceConnectionState); + trace('ice_state', { state: pc.iceConnectionState }); }; + // Diagnostic-only: a periodic health ping and a resume-triggered one, so + // a trace captures exactly what state the connection was in right as the + // page comes back from being backgrounded/locked — never active unless + // trace mode is on (see TRACE_KEY above). + // + // connect() runs again on every reconnect attempt (see _reconnectLoop), + // and each run used to add its own listener/interval on top of the + // previous one without ever removing it — confirmed live: 8 failed + // attempts during one screen lock left 8 duplicate `visibility` trace + // lines firing off the same real event. Disposing of the prior instance + // first is what keeps this to one. + if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; } + if (traceEnabled()) { + const healthPing = async (reason) => { + const before = { + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + channel: this._channel?.readyState, + }; + const start = Date.now(); + try { + await this.ping(8000); + trace('health_ping', { reason, ok: true, rtt_ms: Date.now() - start, ...before }); + } catch (e) { + trace('health_ping', { reason, ok: false, error: String(e && e.message || e), + elapsed_ms: Date.now() - start, ...before }); + } + }; + const onVisibility = () => { + trace('visibility', { + state: document.visibilityState, + pc: this._pc?.connectionState, + ice: this._pc?.iceConnectionState, + channel: this._channel?.readyState, + }); + if (document.visibilityState === 'visible' && this._channel?.readyState === 'open') { + healthPing('resume'); + } + }; + document.addEventListener('visibilitychange', onVisibility); + const healthInterval = setInterval(() => { + if (this._channel?.readyState === 'open') healthPing('interval'); + }, TRACE_PING_INTERVAL_MS); + this._diagCleanup = () => { + document.removeEventListener('visibilitychange', onVisibility); + clearInterval(healthInterval); + }; + } + const offer = await this._pc.createOffer(); await this._pc.setLocalDescription(offer); @@ -426,6 +676,89 @@ class MeshBayTransport { } /** + * Kick off (or join, if one is already running) the automatic reconnect + * after the WebRTC connection is declared unrecoverable. Idempotent: every + * caller racing to reconnect at once — the connectionstatechange handler, + * and any request that lands in the gap _sendAndWait waits out below — + * shares the one attempt instead of piling up parallel handshakes against + * the node. + */ + _reconnect() { + if (this._closed) return Promise.resolve(); + if (!this._reconnectPromise) { + this._reconnectPromise = this._reconnectLoop().finally(() => { + this._reconnectPromise = null; + }); + } + return this._reconnectPromise; + } + + /** + * Redo the signaling handshake from scratch — the only thing that works + * once aiortc has declared a connection "failed": the node discards that + * session the moment it sees the same state (webrtc_server.py's + * on_state_change), so there is no lower-level session left to resume, only + * a fresh one to negotiate. Retries with capped exponential backoff + * (1s, 2s, 4s ... 30s) rather than a fixed number of attempts, because the + * two real causes seen so far — a mobile carrier dropping the NAT mapping + * during screen lock, and the node's own machine being briefly unreachable + * — both resolve on their own eventually, and there is no good moment to + * decide the user would rather see a dead app than keep waiting. + */ + async _reconnectLoop() { + this._reconnectAttempts = 0; + while (!this._closed) { + this._reconnectAttempts += 1; + const delayMs = Math.min(30000, 1000 * 2 ** (this._reconnectAttempts - 1)); + trace('reconnect_wait', { attempt: this._reconnectAttempts, delay_ms: delayMs }); + // Interruptible: _wakeReconnect (fired on visibilitychange → visible) + // resolves this immediately instead of waiting out the rest of a + // backoff that was mostly spent while nothing could succeed anyway. + await new Promise((resolve) => { + const timer = setTimeout(resolve, delayMs); + this._reconnectWakeResolve = () => { clearTimeout(timer); resolve(); }; + }); + this._reconnectWakeResolve = null; + if (this._closed) return; + try { + // Best-effort: these are already unusable, but leaving them wired up + // risks a stray late event from the old pc doing something once a + // new one is in `this._pc` — the `pc === this._pc` guard above closes + // most of that gap, this closes the rest. + try { this._channel && this._channel.close(); } catch { /* already gone */ } + try { this._pc && this._pc.close(); } catch { /* already gone */ } + const args = this._connectArgs; + const token = this._onNeedToken ? await this._onNeedToken() : this._lastToken; + trace('reconnect_attempt', { attempt: this._reconnectAttempts }); + this._inReconnectAttempt = true; + try { + await this.connect(args.nodeId, token, args.groupId, args.gekRaw, + this._sessionKeys, args.bundleKey, args.username, + args.userId, args.joinCode); + } finally { + this._inReconnectAttempt = false; + } + trace('reconnect_ok', { attempt: this._reconnectAttempts }); + console.log('[MeshBay] Reconnected after', this._reconnectAttempts, 'attempt(s)'); + if (this._onReconnected) { + try { this._onReconnected(); } catch (e) { + console.error('[MeshBay] onReconnected handler threw:', e); + } + } + return; + } catch (e) { + trace('reconnect_attempt_failed', { + attempt: this._reconnectAttempts, error: String(e && e.message || e), + }); + console.warn('[MeshBay] Reconnect attempt', this._reconnectAttempts, + 'failed:', e.message); + // Loop again with a longer backoff — closing over `args`/`token` + // freshly next time, in case the token was the actual problem. + } + } + } + + /** * Pair this browser with the node using a one-time code (M3, and the same * substitution as H3). * @@ -1439,7 +1772,39 @@ class MeshBayTransport { get gekRaw() { return this._gekRaw; } + /** + * Give an automatic reconnect already in progress (see _reconnectLoop) a + * bounded chance to land before giving up. + * + * _sendAndWait does this internally for every request that goes through + * it, so most callers never need this directly. It exists for the ones + * that check `transport.connected` themselves before doing anything else — + * music-player.js's fetchTrackBlob is the one this was written for: found + * live throwing "Transport not connected" on the track *after* a + * screen-lock reconnect had already been under way for a while, because + * that check ran, saw `connected` still false, and threw before the + * reconnect it only had to wait a few seconds for got the chance to finish. + * A no-op — returns immediately — when nothing is being reconnected, + * including once one has already succeeded, so it is safe to call + * unconditionally ahead of such a check. + */ + async waitForReconnect(timeoutMs = 6000) { + if (!this._reconnectPromise) return; + await Promise.race([ + this._reconnectPromise.catch(() => {}), + new Promise((r) => setTimeout(r, timeoutMs)), + ]); + } + close() { + // Must be set before pc.close() below: that close() itself can drive the + // pc to "closed" synchronously, and the connectionstatechange handler + // only skips reconnecting because of this flag, not because "closed" is + // absent from its own trigger condition. + this._closed = true; + document.removeEventListener('visibilitychange', this._onVisibilityWake); + this._wakeReconnect(); + if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; } if (this._channel) this._channel.close(); if (this._pc) this._pc.close(); this._connected = false; @@ -1449,13 +1814,25 @@ class MeshBayTransport { // ── Internal ────────────────────────────────────────────────────────────── - _sendAndWait(obj, timeoutMs = 30000) { + async _sendAndWait(obj, timeoutMs = 30000) { + // A reconnect already in flight (see _reconnectLoop) means the channel + // this would send on is the one just declared dead. `_inReconnectAttempt` + // excludes the handshake connect() itself makes while reconnecting — that + // call runs *inside* this same _reconnectPromise, which cannot resolve + // until it returns, so waiting on it here would just be waiting on + // itself for the full 6s, on every step of the handshake, every time. + if (!this._inReconnectAttempt) await this.waitForReconnect(6000); return new Promise((resolve, reject) => { const id = this._seqId++; const timeout = setTimeout(() => { this._pending.delete(id); console.error('[MeshBay] Response timeout for', obj.type, 'after', timeoutMs, 'ms, channel=', this._channel?.readyState); + trace('send_timeout', { + reqType: obj.type, timeoutMs, + pc: this._pc?.connectionState, ice: this._pc?.iceConnectionState, + channel: this._channel?.readyState, + }); reject(new Error('Response timeout')); }, timeoutMs); this._pending.set(id, { |