summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-hub/src/meshbay_hub/static/downloads.js
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-09-08 13:20:29 +0200
committerChristophe Besson <cbesson@gmail.com>2026-09-08 13:20:29 +0200
commit3f2bb22586d3e1aef765149b555ccc8e174ce7eb (patch)
tree7acde4ff7857c5bc2e38b51f40015f7b41823922 /packages/meshbay-hub/src/meshbay_hub/static/downloads.js
parent813d18424ec57963bb56e6f40824a2db0ccce50d (diff)
downloadmeshbay-3f2bb22586d3e1aef765149b555ccc8e174ce7eb.tar.gz
fix(hub): make the streamed download path reliable, and clean up after an abort
On Firefox and Safari the service worker is the only unbounded way to write a download to disk: neither has the File System Access API, and OPFS is not a substitute — measured on Firefox 154, its quota is exactly 10% of the volume's size (389,233,459 bytes of a 3,892,334,592-byte volume, refused to the byte), which a film exceeds. So when this path declines, a large download has nowhere left to go, which makes its reliability a correctness property. Four ways it declined, all of them avoidable: - it was registered inside the first click on Download, so that click paid install, activate and claim while somebody watched a button do nothing; - `_swReady` cached a null for the life of the page. One slow first click left the tab unable to stream anything again, curable only by a reload nobody knew to do. Only a successful controller is remembered now; - control was waited for with a 3 s cap. It is 15 s, and a page that is active but not controlled asks the worker to claim again (`mbdl-claim`) instead of declaring the path unavailable; - a missed navigation gave up at once. It gets a second attempt with a fresh id and iframe, the failed one torn down completely first. Also closes a MessagePort leaked per download, and gives the reason a name (`lastStreamFailure`) so a refusal can say what happened. The timeouts became parameters: the defaults are the production values, no caller passes any, and the tests do not spend a minute waiting. `openTarget` gets an unrelated but adjacent fix, in the same file: it creates the destination with `getFileHandle({create: true})`, so an empty file exists before the first byte, and `abort()` leaves the target untouched — every cancelled download left a 0-byte file behind, and since `freeName` avoids collisions, three cancels left film.mkv, film (2).mkv and film (3).mkv, all empty. Its `abort()` now removes the entry. Safe here and only here, because `freeName` guarantees the name was not taken: the `showSaveFilePicker` path must not do the same, where the person may have picked an existing file whose contents `abort()` correctly preserves. Verified by hand in Chrome. test_streamed_download_reliability.py runs the real module under Node against a stubbed browser — it fails if the null is cached again. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01HCGdheDLxGReuKHga3BtST
Diffstat (limited to 'packages/meshbay-hub/src/meshbay_hub/static/downloads.js')
-rw-r--r--packages/meshbay-hub/src/meshbay_hub/static/downloads.js203
1 files changed, 166 insertions, 37 deletions
diff --git a/packages/meshbay-hub/src/meshbay_hub/static/downloads.js b/packages/meshbay-hub/src/meshbay_hub/static/downloads.js
index 90f88b0..f620e15 100644
--- a/packages/meshbay-hub/src/meshbay_hub/static/downloads.js
+++ b/packages/meshbay-hub/src/meshbay_hub/static/downloads.js
@@ -155,8 +155,28 @@ export async function openTarget(filename) {
};
const name = await freeName(filename, exists);
const handle = await dir.getFileHandle(name, { create: true });
+ const writable = await handle.createWritable();
return {
- writable: await handle.createWritable(),
+ // `abort()` on a FileSystemWritableFileStream discards the swap file and
+ // leaves the target as it was — which here is the empty file
+ // `getFileHandle({create: true})` just made, before a single byte arrived.
+ // So every cancelled or failed download left a 0-byte file behind, and
+ // because `freeName` avoids collisions, three cancels left `film.mkv`,
+ // `film (2).mkv` and `film (3).mkv`, all empty, in the person's folder.
+ //
+ // Removing it is safe *here and only here*: `freeName` guarantees this name
+ // was not taken, so the file being deleted is one we created moments ago
+ // and nothing else. The `showSaveFilePicker` path in file-utils.js must not
+ // do the same — there the person may have picked an existing file, whose
+ // contents `abort()` correctly preserves.
+ writable: {
+ write: (bytes) => writable.write(bytes),
+ close: () => writable.close(),
+ abort: async (reason) => {
+ try { await writable.abort(reason); } catch { /* already gone */ }
+ try { await dir.removeEntry(name); } catch { /* already gone */ }
+ },
+ },
name,
// Reading it back is the only way a page can "open" a file it wrote: hand
// the bytes to a tab and let the browser decide what to do with them. No
@@ -180,40 +200,114 @@ export const BLOB_LIMIT = 512 * 1024 * 1024;
// ── Streaming to disk without the File System Access API ────────────────────
const SW_PATH = '/sw.js';
-let _swReady = null;
+
+// On Firefox and Safari this worker is not a nicety, it is the only unbounded
+// way to write a download to disk: the File System Access API does not exist
+// there, and OPFS is capped at 10% of the volume's size (measured on Firefox
+// 154: 389,233,459 bytes on a 3,892,334,592-byte volume, refused to the byte),
+// which a film can exceed. Everything below exists to make sure this path is
+// available when it is needed, because there is nothing underneath it.
+//
+// How long to wait for this page to become *controlled*. Generous on purpose:
+// the cost of waiting is a spinner, and the cost of giving up is a download
+// this browser then cannot do at all.
+const SW_CONTROL_BUDGET_MS = 15000;
+// How long to wait for the worker to confirm it answered the iframe.
+const SW_SERVED_BUDGET_MS = 15000;
+// A transient miss gets a second go with a fresh id and a fresh iframe.
+const SW_ATTEMPTS = 2;
+
+// Holds a *successful* controller, or an in-flight attempt. Never a failure —
+// see serviceWorker(). The previous version cached the rejected/null result
+// for the life of the page, so one slow first click (a cold worker, a busy
+// phone) left the tab unable to stream anything ever again, with no way back
+// but a reload nobody knew to do.
+let _swPromise = null;
+let _lastFailure = '';
+
+/** Why the streamed path last declined, for a message worth reading. */
+export function lastStreamFailure() { return _lastFailure; }
export const STREAMS_VIA_SW = typeof window !== 'undefined'
&& 'serviceWorker' in navigator
&& typeof TransformStream === 'function'
&& window.isSecureContext;
-async function serviceWorker() {
- if (!STREAMS_VIA_SW) return null;
- if (!_swReady) {
- _swReady = navigator.serviceWorker.register(SW_PATH, { scope: '/' })
- .then(() => navigator.serviceWorker.ready)
- .then(async (reg) => {
- // `reg.active` is not enough. A worker can be active while this page is
- // still uncontrolled, and an uncontrolled page's requests are never
- // handed to its fetch handler — so the worker would take our stream and
- // then never be asked for it. The iframe would 404, nothing would read
- // the stream, and the first write() would block for good: a download
- // stuck at one chunk.
- if (navigator.serviceWorker.controller) return navigator.serviceWorker.controller;
- // sw.js claims clients on activate, so control usually arrives within a
- // tick of registration. Wait briefly rather than give up at once.
- return await new Promise((resolve) => {
- const done = () => resolve(navigator.serviceWorker.controller || null);
- navigator.serviceWorker.addEventListener('controllerchange', done, { once: true });
- setTimeout(done, 3000);
- });
- })
- .catch(err => {
- console.warn('[MeshBay] service worker unavailable:', err.message);
- return null;
- });
+/** Resolves with the controller, or null once `budgetMs` is spent. */
+function _awaitControl(budgetMs) {
+ if (navigator.serviceWorker.controller) {
+ return Promise.resolve(navigator.serviceWorker.controller);
+ }
+ return new Promise((resolve) => {
+ let timer = 0;
+ const done = () => {
+ clearTimeout(timer);
+ navigator.serviceWorker.removeEventListener('controllerchange', done);
+ resolve(navigator.serviceWorker.controller || null);
+ };
+ navigator.serviceWorker.addEventListener('controllerchange', done);
+ timer = setTimeout(done, budgetMs);
+ });
+}
+
+async function _claimController(budgetMs) {
+ const reg = await navigator.serviceWorker.register(SW_PATH, { scope: '/' });
+ // `ready` resolves on an *active* registration; being active is not being in
+ // control. An uncontrolled page's requests never reach the fetch handler, so
+ // the worker would take our stream and never be asked for it — the download
+ // then freezes after exactly one chunk, which is how this was found.
+ await navigator.serviceWorker.ready;
+ const controller = await _awaitControl(budgetMs);
+ if (controller) return controller;
+ // Active but not controlling after the whole budget. `sw.js` calls
+ // `clients.claim()` on activate, so this is rare; when it happens the page
+ // was loaded before any worker existed and the claim was missed. Ask the
+ // active worker to claim again rather than declare the path unavailable.
+ if (reg.active) {
+ try { reg.active.postMessage({ type: 'mbdl-claim' }); } catch { /* gone */ }
+ return await _awaitControl(2000);
+ }
+ return null;
+}
+
+/**
+ * Register the worker and get this page controlled, now.
+ *
+ * Called at application start, not at the first download. Registration used to
+ * happen inside the first click, so that click paid install, activate and claim
+ * while somebody watched a button do nothing — and if the claim did not land
+ * inside the budget, the download fell through to a path that cannot hold a
+ * film. By the time anyone clicks anything, this has long since finished.
+ *
+ * Fire-and-forget by design: nothing waits on it, and a failure here is not
+ * fatal because `serviceWorker()` will simply try again.
+ */
+export function primeServiceWorker() {
+ if (!STREAMS_VIA_SW) return;
+ serviceWorker().catch(() => {});
+}
+
+async function serviceWorker(controlMs = SW_CONTROL_BUDGET_MS) {
+ if (!STREAMS_VIA_SW) {
+ _lastFailure = 'no service worker support in this browser';
+ return null;
}
- return _swReady;
+ if (navigator.serviceWorker.controller) return navigator.serviceWorker.controller;
+ if (!_swPromise) {
+ _swPromise = _claimController(controlMs).catch((err) => {
+ _lastFailure = 'service worker registration failed: ' + err.message;
+ console.warn('[MeshBay]', _lastFailure);
+ return null;
+ });
+ }
+ const controller = await _swPromise;
+ if (!controller) {
+ // Not remembered. The next attempt starts from scratch, which is the whole
+ // point: these failures are transient far more often than they are final.
+ _swPromise = null;
+ if (!_lastFailure) _lastFailure = 'the page did not come under the worker’s control';
+ }
+ return controller;
}
/**
@@ -229,8 +323,29 @@ async function serviceWorker() {
* Returns {writable, name} shaped like the File System Access one, or null if
* this browser cannot do it either.
*/
-export async function openStreamedDownload(filename, size = 0) {
- const worker = await serviceWorker();
+export async function openStreamedDownload(filename, size = 0, {
+ controlMs = SW_CONTROL_BUDGET_MS,
+ servedMs = SW_SERVED_BUDGET_MS,
+ attempts = SW_ATTEMPTS,
+} = {}) {
+ for (let attempt = 1; attempt <= attempts; attempt++) {
+ const target = await _attemptStreamedDownload(
+ filename, size, attempt, controlMs, servedMs);
+ if (target) return target;
+ // A miss is usually the worker having been asleep or the navigation losing
+ // a race, not this browser being unable. Falling through on the first miss
+ // is what sent large downloads to the in-memory floor.
+ if (attempt < attempts) {
+ console.warn(`[MeshBay] streamed download attempt ${attempt} missed `
+ + `(${_lastFailure}); retrying`);
+ }
+ }
+ return null;
+}
+
+async function _attemptStreamedDownload(filename, size, attempt,
+ controlMs, servedMs) {
+ const worker = await serviceWorker(controlMs);
if (!worker) return null;
const id = `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`;
@@ -252,8 +367,11 @@ export async function openStreamedDownload(filename, size = 0) {
[readable, chan.port2]);
} catch (err) {
// Transferable streams are what makes the backpressure work; without them
- // this would be a memory buffer wearing a stream's clothes.
- console.warn('[MeshBay] streams cannot be transferred here:', err.message);
+ // this would be a memory buffer wearing a stream's clothes. This one is
+ // final rather than transient — a browser does not grow the capability
+ // between two attempts — so it is reported as such.
+ _lastFailure = 'this browser cannot transfer a stream to the worker: ' + err.message;
+ console.warn('[MeshBay]', _lastFailure);
return null;
}
@@ -264,19 +382,30 @@ export async function openStreamedDownload(filename, size = 0) {
const answered = await Promise.race([
serving,
- new Promise((r) => setTimeout(() => r(false), 8000)),
+ new Promise((r) => setTimeout(() => r(false), servedMs)),
]);
if (!answered) {
// Some browsers refuse a download started from a hidden iframe, and an
- // uncontrolled page never reaches the worker at all. Say so and let the
- // caller fall back rather than hand back a sink nothing drains.
- console.warn('[MeshBay] the service worker never served the download; '
- + 'falling back');
+ // uncontrolled page never reaches the worker at all. Tear this attempt
+ // down completely — the stream, the port and the frame — so a retry starts
+ // clean rather than leaving a half-open sink behind.
+ _lastFailure = `the worker did not answer the download within `
+ + `${servedMs / 1000}s (attempt ${attempt})`;
+ console.warn('[MeshBay]', _lastFailure);
frame.remove();
+ try { chan.port1.close(); } catch { /* already gone */ }
try { await writable.abort('not served'); } catch { /* already gone */ }
return null;
}
+ _lastFailure = '';
+ // The port has delivered the one message it exists for. Closing it matters:
+ // an open MessagePort is a live handle, and one was leaked per download for
+ // the life of the page. (It is also what hung the Node harness in
+ // test_streamed_download_reliability.py — there the leak is a process that
+ // never exits, which is the same defect wearing a louder symptom.)
+ try { chan.port1.close(); } catch { /* already gone */ }
+
const writer = writable.getWriter();
return {
name: filename,