aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-hub/src/meshbay_hub/static/transport.js
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-08-26 15:09:20 +0200
committerChristophe Besson <cbesson@gmail.com>2026-08-26 15:09:20 +0200
commitdb7f81fd08742847f3ebea061c75530b7b31b934 (patch)
tree14b308686860eea0a9c12c47c39f8a68a1bba376 /packages/meshbay-hub/src/meshbay_hub/static/transport.js
parent7126fd3c265ba75d77b449bfd0f83f5f3e584b74 (diff)
parent59d9f50bf41b2b38b7c95f8da52b698dff923c2d (diff)
downloadmeshbay-db7f81fd08742847f3ebea061c75530b7b31b934.tar.gz
Merge branch 'debug/webrtc-lock-resume'
WebRTC transport dies silently after an extended mobile screen lock (confirmed live via client-side trace + node logs): ICE goes disconnected -> failed within ~10s of each other on both ends, but the DataChannel's readyState stays "open" throughout, so nothing failed fast — every request just sat out its own timeout, matching the reported symptom (poster spinners, blocked chat, dead new streams, stuck music). - Automatic reconnect on WebRTC "failed": capped exponential backoff, redoes the full signaling handshake, wakes immediately on visibilitychange instead of waiting out a throttled backoff timer. - Fixed two real bugs the reconnect work exposed: the signaling POST to the hub kept using the token captured at construction, never the fresh one fetched per reconnect attempt (401 loop, no possible recovery); and connect() re-armed a diagnostic listener/interval on every attempt without disposing of the previous one. - pipelinedDownload retries a lost chunk instead of aborting the whole transfer — covers Files downloads, video poster/thumbnail fetches, and music-player.js's blob-based track download. - music-player.js: don't throw "Transport not connected" while a reconnect is already landing (waitForReconnect); prefetch depth now adapts to network type (5 tracks ahead on Wi-Fi, 3 on cellular or unrecognized — Firefox/Safari included, where the detection API is simply absent). - video-player.js: onReconnected reissues the existing seek-to-current-time path, so a mid-stream reconnect looks like an ordinary seek rather than a dead player; holds a Screen Wake Lock unconditionally while open. - New opt-in (off by default) user preference: keep the screen on during audio playback, for whoever wants to trade battery for sidestepping the screen-lock gap entirely — off by default because the ordinary expectation (matching Spotify/Deezer) is that the phone locks on its own while listening. - hub: /app and / now serve Cache-Control: no-store — the SPA shell had no cache header at all, so a browser that cached it heuristically could keep re-serving an old build (old ASSET_V, old JS) through any number of reloads or pull-to-refreshes. Verified against real production use across many rounds (demo groups, actual mobile screen-lock testing) rather than synthetic reproduction alone. 430 hub tests + 650 node tests passing throughout.
Diffstat (limited to 'packages/meshbay-hub/src/meshbay_hub/static/transport.js')
-rw-r--r--packages/meshbay-hub/src/meshbay_hub/static/transport.js387
1 files changed, 382 insertions, 5 deletions
diff --git a/packages/meshbay-hub/src/meshbay_hub/static/transport.js b/packages/meshbay-hub/src/meshbay_hub/static/transport.js
index 7535594..9f85ee1 100644
--- a/packages/meshbay-hub/src/meshbay_hub/static/transport.js
+++ b/packages/meshbay-hub/src/meshbay_hub/static/transport.js
@@ -81,6 +81,103 @@ const ADMIN_OP_TYPES = new Set([
'group_detach', 'invite_create',
]);
+// ── Diagnostic trace (opt-in, off by default) ───────────────────────────────
+// Ring buffer of transport health events (connection/ICE/DataChannel state
+// transitions, request timeouts, visibility changes, periodic health pings),
+// persisted to localStorage so a connection that gets stuck can be inspected
+// after the fact — the field case this exists for is a phone with no
+// devtools attached. Added while chasing a report of the transport going
+// unresponsive after a mobile screen lock of several minutes; kept in the
+// tree afterward rather than ripped out, since the next hard-to-reproduce
+// connection bug will want the same thing and it costs nothing while off.
+//
+// Enable once by opening the app with ?trace=1 in the URL — this persists in
+// localStorage, so every later visit stays in trace mode until ?trace=0
+// clears it. Read the log back at any time by navigating to #mb-debug (e.g.
+// https://meshbay.org/app/#mb-debug), which replaces the page with a plain
+// text dump — no devtools required.
+const TRACE_KEY = 'mb_trace';
+const TRACE_LOG_KEY = 'mb_trace_log';
+const TRACE_MAX = 500;
+// How often to probe the channel with a ping while trace mode is on — purely
+// diagnostic (to see when a health check starts failing), not a keepalive:
+// must stay opt-in, never run by default.
+const TRACE_PING_INTERVAL_MS = 25000;
+
+(function _initTraceFlag() {
+ try {
+ const params = new URLSearchParams(location.search);
+ if (params.has('trace')) {
+ if (params.get('trace') === '0') localStorage.removeItem(TRACE_KEY);
+ else localStorage.setItem(TRACE_KEY, '1');
+ }
+ } catch { /* localStorage unavailable (private mode, etc.) — trace stays off */ }
+})();
+
+function traceEnabled() {
+ try { return localStorage.getItem(TRACE_KEY) === '1'; } catch { return false; }
+}
+
+function trace(event, data) {
+ if (!traceEnabled()) return;
+ try {
+ const buf = JSON.parse(localStorage.getItem(TRACE_LOG_KEY) || '[]');
+ buf.push({ t: new Date().toISOString(), event, ...data });
+ while (buf.length > TRACE_MAX) buf.shift();
+ localStorage.setItem(TRACE_LOG_KEY, JSON.stringify(buf));
+ } catch { /* storage full or unavailable — tracing is best-effort */ }
+}
+
+window.MeshBayTrace = {
+ enabled: traceEnabled,
+ dump() {
+ try { return JSON.parse(localStorage.getItem(TRACE_LOG_KEY) || '[]'); } catch { return []; }
+ },
+ clear() { try { localStorage.removeItem(TRACE_LOG_KEY); } catch { /* ignore */ } },
+};
+
+function _showTraceView() {
+ {
+ const renderTraceView = () => {
+ const log = window.MeshBayTrace.dump();
+ const text = JSON.stringify(log, null, 2);
+ document.body.innerHTML = '';
+ document.title = 'MeshBay — Diagnostic';
+ const bar = document.createElement('div');
+ bar.style.cssText = 'font-family:monospace;padding:8px;';
+ const copyBtn = document.createElement('button');
+ copyBtn.textContent = 'Copier';
+ copyBtn.onclick = () => { navigator.clipboard.writeText(text).catch(() => {}); };
+ const clearBtn = document.createElement('button');
+ clearBtn.textContent = 'Vider';
+ clearBtn.onclick = () => { window.MeshBayTrace.clear(); renderTraceView(); };
+ const refreshBtn = document.createElement('button');
+ refreshBtn.textContent = 'Rafraîchir';
+ refreshBtn.onclick = renderTraceView;
+ const info = document.createElement('span');
+ info.textContent = ` — ${log.length} évènement(s) — trace ${traceEnabled() ? 'active' : 'inactive'}`;
+ info.style.marginLeft = '8px';
+ bar.append(copyBtn, clearBtn, refreshBtn, info);
+ const pre = document.createElement('pre');
+ pre.style.cssText = 'font-family:monospace;font-size:11px;white-space:pre-wrap;'
+ + 'word-break:break-all;padding:8px;';
+ pre.textContent = text;
+ document.body.append(bar, pre);
+ };
+ renderTraceView();
+ }
+}
+
+// Fragment-only URL changes (typing #mb-debug into an already-loaded page,
+// or a link to it) do not reload the document, so DOMContentLoaded alone
+// would miss them — hashchange is what a same-document navigation fires.
+if (location.hash === '#mb-debug') {
+ document.addEventListener('DOMContentLoaded', _showTraceView);
+}
+window.addEventListener('hashchange', () => {
+ if (location.hash === '#mb-debug') _showTraceView();
+});
+
const JOIN_REFUSALS = {
code_required: 'This node does not know this browser yet. Ask the node operator '
+ 'for a pairing code (meshbay-node operator pair).',
@@ -117,6 +214,50 @@ class MeshBayTransport {
// several uploads may be in flight at once and their acks interleave; the
// node names the file in every one.
this._uploaders = new Map();
+ // Set once close() runs — stops the automatic reconnect from firing on a
+ // connection the caller tore down on purpose (leaving the group, page
+ // unload), which would otherwise race back in right as everything else
+ // is being torn down.
+ this._closed = false;
+ // The arguments connect() was last given, minus the token (refreshed at
+ // reconnect time — see onNeedToken) and sessionKeys (kept live on `this`,
+ // since a reconnect must reuse the identity connect() settled on, not
+ // whatever the very first caller passed in — see _reconnectLoop).
+ this._connectArgs = null;
+ this._lastToken = null;
+ this._reconnectPromise = null;
+ this._reconnectAttempts = 0;
+ // True only for the duration of the connect() call _reconnectLoop makes
+ // to actually retry — as opposed to the backoff delay around it, which
+ // is most of _reconnectPromise's lifetime. Needed because that connect()
+ // call sends its own handshake through _sendAndWait, which would
+ // otherwise see the very _reconnectPromise it is running inside of as
+ // "a reconnect to wait for" and stall every handshake step for the full
+ // 6s gate below before ever sending it.
+ this._inReconnectAttempt = false;
+ this._onReconnected = null;
+ this._onNeedToken = null;
+ // Cuts the backoff wait short the moment the page is foregrounded again —
+ // found live to matter: a screen lock throttles the tab's own timers
+ // along with everything else, so a backoff already counting down when the
+ // phone locked can run for minutes of *wall clock* past its nominal delay
+ // before it next gets to run at all. Set once, here, rather than inside
+ // connect() like the diagnostic listener above it — this one has to
+ // survive every reconnect attempt, not restart with each one.
+ this._reconnectWakeResolve = null;
+ this._onVisibilityWake = () => {
+ if (document.visibilityState === 'visible') this._wakeReconnect();
+ };
+ document.addEventListener('visibilitychange', this._onVisibilityWake);
+ }
+
+ /** Cuts short a reconnect currently backing off (see _reconnectLoop). A
+ * no-op when nothing is waiting, so this is safe to call unconditionally. */
+ _wakeReconnect() {
+ if (this._reconnectWakeResolve) {
+ this._reconnectWakeResolve();
+ this._reconnectWakeResolve = null;
+ }
}
get connected() { return this._connected; }
@@ -138,6 +279,17 @@ class MeshBayTransport {
set onMusicbrainzConfig(fn) { this._onMusicbrainzConfig = fn; }
set onMusicbrainzEnabled(fn) { this._onMusicbrainzEnabled = fn; }
set onIndexProgress(fn) { this._onIndexProgress = fn; }
+ // Fired once an automatic reconnect (see _reconnectLoop) lands a fresh
+ // handshake, so a consumer with something mid-flight on the old channel —
+ // today only the video player — can pick back up rather than sit dead.
+ set onReconnected(fn) { this._onReconnected = fn; }
+ // Reconnecting redoes the handshake, which needs a JWT that may have gone
+ // stale while the connection was down for minutes. Without this the
+ // reconnect resends whatever token the original connect() call captured,
+ // which the node's clock-skew check (stale_request) or plain expiry can
+ // by then have already invalidated. Set to whatever the caller uses to
+ // refresh the hub session token (see group-page.js's ensureFreshToken).
+ set onNeedToken(fn) { this._onNeedToken = fn; }
get sessionKeys() { return this._sessionKeys; }
@@ -147,6 +299,21 @@ class MeshBayTransport {
async connect(nodeId, jwtToken, groupId, gekRaw, sessionKeys, bundleKey, username,
userId, joinCode) {
+ // Remembered for _reconnectLoop, which calls connect() again with these
+ // same values (plus a freshly-fetched token and the identity connect()
+ // itself settles on below) after the WebRTC connection is declared
+ // "failed" — see the pc.onconnectionstatechange handler further down.
+ this._connectArgs = { nodeId, groupId, gekRaw, bundleKey, username, userId, joinCode };
+ this._lastToken = jwtToken;
+ // The constructor sets this once from whatever token the caller had at
+ // the time — and the signaling POST below reads *this*, not `jwtToken`.
+ // A reconnect passes a freshly-fetched `jwtToken` (see onNeedToken) but
+ // that never reached here before, so the signaling call kept using the
+ // original token no matter how many minutes had passed or how many
+ // reconnect attempts fetched a new one — confirmed live: every attempt
+ // failed "Signaling failed: 401 Invalid or expired token" in a loop,
+ // never actually trying the fresh token connect() had just been handed.
+ this._accessToken = jwtToken;
this._gekRaw = gekRaw || null;
this._sessionKeys = sessionKeys || null;
this._bundleKey = bundleKey || null;
@@ -168,6 +335,7 @@ class MeshBayTransport {
this._channel.onopen = () => {
clearTimeout(timeout);
this._connected = true;
+ trace('channel_open', {});
resolve();
};
});
@@ -175,6 +343,11 @@ class MeshBayTransport {
this._channel.onmessage = (event) => this._onMessage(event.data);
this._channel.onclose = (ev) => {
console.warn('[MeshBay] DataChannel closed', this._channel?.readyState, ev);
+ trace('channel_close', {
+ readyState: this._channel?.readyState,
+ pc: this._pc?.connectionState,
+ ice: this._pc?.iceConnectionState,
+ });
this._connected = false;
if (channelReject) channelReject(new Error('DataChannel closed'));
for (const [, p] of this._pending) p.reject(new Error('DataChannel closed'));
@@ -182,16 +355,93 @@ class MeshBayTransport {
};
this._channel.onerror = (ev) => {
console.error('[MeshBay] DataChannel error', ev);
+ trace('channel_error', {
+ pc: this._pc?.connectionState,
+ ice: this._pc?.iceConnectionState,
+ });
if (channelReject) channelReject(new Error('DataChannel error'));
};
- this._pc.onconnectionstatechange = () => {
- console.log('[MeshBay] PC state:', this._pc.connectionState);
+ // Captured locally rather than read back through `this._pc`: once a
+ // reconnect replaces it, a late event from this (by then orphaned) pc
+ // must still be judged against the pc it actually came from, not
+ // whatever is current — the `pc === this._pc` check below is what that
+ // buys.
+ const pc = this._pc;
+ pc.onconnectionstatechange = () => {
+ console.log('[MeshBay] PC state:', pc.connectionState);
+ trace('pc_state', { state: pc.connectionState });
+ // "failed" is ICE's own verdict that nothing here will recover on its
+ // own (unlike a transient "disconnected", which often clears itself) —
+ // confirmed live: mobile screen lock for several minutes reliably
+ // produces disconnected → failed about 10s apart, on both ends, and
+ // nothing today ever moves past that without a full page reload.
+ // `channel.readyState` is no help distinguishing this: it was observed
+ // staying "open" throughout, so every send from here on would simply
+ // sit out its own timeout instead of failing fast.
+ if (pc.connectionState === 'failed' && pc === this._pc && !this._closed) {
+ this._connected = false;
+ this._reconnect();
+ const err = new Error('WebRTC connection lost');
+ err.name = 'TransportLostError';
+ for (const [, p] of this._pending) p.reject(err);
+ this._pending.clear();
+ }
};
- this._pc.oniceconnectionstatechange = () => {
- console.log('[MeshBay] ICE state:', this._pc.iceConnectionState);
+ pc.oniceconnectionstatechange = () => {
+ console.log('[MeshBay] ICE state:', pc.iceConnectionState);
+ trace('ice_state', { state: pc.iceConnectionState });
};
+ // Diagnostic-only: a periodic health ping and a resume-triggered one, so
+ // a trace captures exactly what state the connection was in right as the
+ // page comes back from being backgrounded/locked — never active unless
+ // trace mode is on (see TRACE_KEY above).
+ //
+ // connect() runs again on every reconnect attempt (see _reconnectLoop),
+ // and each run used to add its own listener/interval on top of the
+ // previous one without ever removing it — confirmed live: 8 failed
+ // attempts during one screen lock left 8 duplicate `visibility` trace
+ // lines firing off the same real event. Disposing of the prior instance
+ // first is what keeps this to one.
+ if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; }
+ if (traceEnabled()) {
+ const healthPing = async (reason) => {
+ const before = {
+ pc: this._pc?.connectionState,
+ ice: this._pc?.iceConnectionState,
+ channel: this._channel?.readyState,
+ };
+ const start = Date.now();
+ try {
+ await this.ping(8000);
+ trace('health_ping', { reason, ok: true, rtt_ms: Date.now() - start, ...before });
+ } catch (e) {
+ trace('health_ping', { reason, ok: false, error: String(e && e.message || e),
+ elapsed_ms: Date.now() - start, ...before });
+ }
+ };
+ const onVisibility = () => {
+ trace('visibility', {
+ state: document.visibilityState,
+ pc: this._pc?.connectionState,
+ ice: this._pc?.iceConnectionState,
+ channel: this._channel?.readyState,
+ });
+ if (document.visibilityState === 'visible' && this._channel?.readyState === 'open') {
+ healthPing('resume');
+ }
+ };
+ document.addEventListener('visibilitychange', onVisibility);
+ const healthInterval = setInterval(() => {
+ if (this._channel?.readyState === 'open') healthPing('interval');
+ }, TRACE_PING_INTERVAL_MS);
+ this._diagCleanup = () => {
+ document.removeEventListener('visibilitychange', onVisibility);
+ clearInterval(healthInterval);
+ };
+ }
+
const offer = await this._pc.createOffer();
await this._pc.setLocalDescription(offer);
@@ -426,6 +676,89 @@ class MeshBayTransport {
}
/**
+ * Kick off (or join, if one is already running) the automatic reconnect
+ * after the WebRTC connection is declared unrecoverable. Idempotent: every
+ * caller racing to reconnect at once — the connectionstatechange handler,
+ * and any request that lands in the gap _sendAndWait waits out below —
+ * shares the one attempt instead of piling up parallel handshakes against
+ * the node.
+ */
+ _reconnect() {
+ if (this._closed) return Promise.resolve();
+ if (!this._reconnectPromise) {
+ this._reconnectPromise = this._reconnectLoop().finally(() => {
+ this._reconnectPromise = null;
+ });
+ }
+ return this._reconnectPromise;
+ }
+
+ /**
+ * Redo the signaling handshake from scratch — the only thing that works
+ * once aiortc has declared a connection "failed": the node discards that
+ * session the moment it sees the same state (webrtc_server.py's
+ * on_state_change), so there is no lower-level session left to resume, only
+ * a fresh one to negotiate. Retries with capped exponential backoff
+ * (1s, 2s, 4s ... 30s) rather than a fixed number of attempts, because the
+ * two real causes seen so far — a mobile carrier dropping the NAT mapping
+ * during screen lock, and the node's own machine being briefly unreachable
+ * — both resolve on their own eventually, and there is no good moment to
+ * decide the user would rather see a dead app than keep waiting.
+ */
+ async _reconnectLoop() {
+ this._reconnectAttempts = 0;
+ while (!this._closed) {
+ this._reconnectAttempts += 1;
+ const delayMs = Math.min(30000, 1000 * 2 ** (this._reconnectAttempts - 1));
+ trace('reconnect_wait', { attempt: this._reconnectAttempts, delay_ms: delayMs });
+ // Interruptible: _wakeReconnect (fired on visibilitychange → visible)
+ // resolves this immediately instead of waiting out the rest of a
+ // backoff that was mostly spent while nothing could succeed anyway.
+ await new Promise((resolve) => {
+ const timer = setTimeout(resolve, delayMs);
+ this._reconnectWakeResolve = () => { clearTimeout(timer); resolve(); };
+ });
+ this._reconnectWakeResolve = null;
+ if (this._closed) return;
+ try {
+ // Best-effort: these are already unusable, but leaving them wired up
+ // risks a stray late event from the old pc doing something once a
+ // new one is in `this._pc` — the `pc === this._pc` guard above closes
+ // most of that gap, this closes the rest.
+ try { this._channel && this._channel.close(); } catch { /* already gone */ }
+ try { this._pc && this._pc.close(); } catch { /* already gone */ }
+ const args = this._connectArgs;
+ const token = this._onNeedToken ? await this._onNeedToken() : this._lastToken;
+ trace('reconnect_attempt', { attempt: this._reconnectAttempts });
+ this._inReconnectAttempt = true;
+ try {
+ await this.connect(args.nodeId, token, args.groupId, args.gekRaw,
+ this._sessionKeys, args.bundleKey, args.username,
+ args.userId, args.joinCode);
+ } finally {
+ this._inReconnectAttempt = false;
+ }
+ trace('reconnect_ok', { attempt: this._reconnectAttempts });
+ console.log('[MeshBay] Reconnected after', this._reconnectAttempts, 'attempt(s)');
+ if (this._onReconnected) {
+ try { this._onReconnected(); } catch (e) {
+ console.error('[MeshBay] onReconnected handler threw:', e);
+ }
+ }
+ return;
+ } catch (e) {
+ trace('reconnect_attempt_failed', {
+ attempt: this._reconnectAttempts, error: String(e && e.message || e),
+ });
+ console.warn('[MeshBay] Reconnect attempt', this._reconnectAttempts,
+ 'failed:', e.message);
+ // Loop again with a longer backoff — closing over `args`/`token`
+ // freshly next time, in case the token was the actual problem.
+ }
+ }
+ }
+
+ /**
* Pair this browser with the node using a one-time code (M3, and the same
* substitution as H3).
*
@@ -1439,7 +1772,39 @@ class MeshBayTransport {
get gekRaw() { return this._gekRaw; }
+ /**
+ * Give an automatic reconnect already in progress (see _reconnectLoop) a
+ * bounded chance to land before giving up.
+ *
+ * _sendAndWait does this internally for every request that goes through
+ * it, so most callers never need this directly. It exists for the ones
+ * that check `transport.connected` themselves before doing anything else —
+ * music-player.js's fetchTrackBlob is the one this was written for: found
+ * live throwing "Transport not connected" on the track *after* a
+ * screen-lock reconnect had already been under way for a while, because
+ * that check ran, saw `connected` still false, and threw before the
+ * reconnect it only had to wait a few seconds for got the chance to finish.
+ * A no-op — returns immediately — when nothing is being reconnected,
+ * including once one has already succeeded, so it is safe to call
+ * unconditionally ahead of such a check.
+ */
+ async waitForReconnect(timeoutMs = 6000) {
+ if (!this._reconnectPromise) return;
+ await Promise.race([
+ this._reconnectPromise.catch(() => {}),
+ new Promise((r) => setTimeout(r, timeoutMs)),
+ ]);
+ }
+
close() {
+ // Must be set before pc.close() below: that close() itself can drive the
+ // pc to "closed" synchronously, and the connectionstatechange handler
+ // only skips reconnecting because of this flag, not because "closed" is
+ // absent from its own trigger condition.
+ this._closed = true;
+ document.removeEventListener('visibilitychange', this._onVisibilityWake);
+ this._wakeReconnect();
+ if (this._diagCleanup) { this._diagCleanup(); this._diagCleanup = null; }
if (this._channel) this._channel.close();
if (this._pc) this._pc.close();
this._connected = false;
@@ -1449,13 +1814,25 @@ class MeshBayTransport {
// ── Internal ──────────────────────────────────────────────────────────────
- _sendAndWait(obj, timeoutMs = 30000) {
+ async _sendAndWait(obj, timeoutMs = 30000) {
+ // A reconnect already in flight (see _reconnectLoop) means the channel
+ // this would send on is the one just declared dead. `_inReconnectAttempt`
+ // excludes the handshake connect() itself makes while reconnecting — that
+ // call runs *inside* this same _reconnectPromise, which cannot resolve
+ // until it returns, so waiting on it here would just be waiting on
+ // itself for the full 6s, on every step of the handshake, every time.
+ if (!this._inReconnectAttempt) await this.waitForReconnect(6000);
return new Promise((resolve, reject) => {
const id = this._seqId++;
const timeout = setTimeout(() => {
this._pending.delete(id);
console.error('[MeshBay] Response timeout for', obj.type,
'after', timeoutMs, 'ms, channel=', this._channel?.readyState);
+ trace('send_timeout', {
+ reqType: obj.type, timeoutMs,
+ pc: this._pc?.connectionState, ice: this._pc?.iceConnectionState,
+ channel: this._channel?.readyState,
+ });
reject(new Error('Response timeout'));
}, timeoutMs);
this._pending.set(id, {