1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
|
"""
The service-worker download path, which on Firefox and Safari is the only
unbounded way to write a file to disk.
Neither of those browsers has the File System Access API, and OPFS is not a
substitute: measured on Firefox 154, its quota is exactly 10% of the volume's
size (389,233,459 bytes on a 3,892,334,592-byte volume, refused to the byte),
which a film exceeds. So when this path declines, a large download has nowhere
left to go — there is no floor under it that can hold a film. That is what makes
its reliability a correctness property rather than a nicety.
The real module is imported under Node with the browser pieces it reaches
stubbed — `navigator.serviceWorker`, a document that "navigates" an iframe, and
Node's own TransformStream and MessageChannel, which are the real ones. What is
modelled is the environment; `serviceWorker()` and `openStreamedDownload()` are
executed, never reimplemented.
Three failures are pinned, all of which shipped:
- registration happened inside the first click, so that click paid install,
activate and claim while somebody watched a button do nothing;
- a null result was cached for the life of the page, so one slow first click
left the tab unable to stream anything again, curable only by a reload
nobody knew to do;
- one missed navigation fell straight through instead of retrying.
"""
import json
import shutil
import subprocess
from pathlib import Path
import pytest
STATIC = Path(__file__).resolve().parents[1] / "src" / "meshbay_hub" / "static"
DOWNLOADS = STATIC / "downloads.js"
pytestmark = pytest.mark.skipif(
shutil.which("node") is None or not DOWNLOADS.exists(),
reason="node or the SPA sources are not available")
# The stub browser. `plan` decides how the fake worker behaves, so one harness
# covers every case below.
PRELUDE = """
const store = new Map();
globalThis.localStorage = {
getItem: k => (store.has(k) ? store.get(k) : null),
setItem: (k, v) => store.set(k, String(v)),
removeItem: k => store.delete(k),
};
const PLAN = %(plan)s;
const log = { registers: 0, claims: 0, navigations: 0, served: 0,
unregisters: 0 };
// The worker as the page sees it: something with postMessage. It answers a
// navigation by posting mbdl-serving back on the port it was handed, which is
// exactly the confirmation the real sw.js sends from its fetch handler.
let controller = null;
const pendingByFrame = new Map();
const makeController = () => ({
postMessage: (msg, transfer) => {
if (msg.type === 'mbdl-claim') {
log.claims += 1;
// A worker that actually claims when asked, which is what sw.js does.
if (PLAN.controlOnClaim) {
controller = makeController();
for (const fn of listeners) fn();
}
return;
}
if (msg.type !== 'mbdl') return;
pendingByFrame.set('/_mbdl/' + msg.id, msg.port);
},
});
const listeners = new Set();
// `globalThis.navigator` is read-only from Node 22 -- assigning to it is the
// mistake CLAUDE.md already records against test_locales.py. Define it.
Object.defineProperty(globalThis, 'navigator', {
configurable: true,
value: {
serviceWorker: {
get controller() { return controller; },
register: async () => {
log.registers += 1;
if (PLAN.registerThrows) throw new Error('registration blocked');
// A registration that never answers at all. Distinct from one that
// rejects: nothing is reported, nothing fails, the caller just waits.
if (PLAN.registerHangs) await new Promise(() => {});
// A worker that only becomes installable once the stuck registration
// has been thrown away -- the browser this was reported from.
const healed = PLAN.activeAfterUnregister && log.unregisters > 0;
if (PLAN.controlAfterMs !== null || healed) {
setTimeout(() => {
controller = makeController();
for (const fn of listeners) fn();
}, healed ? 0 : PLAN.controlAfterMs);
}
return {
active: (PLAN.active || healed) ? makeController() : null,
unregister: async () => { log.unregisters += 1; return true; },
};
},
// `register()` resolves as soon as the registration object exists, with
// nothing but an installing worker; `ready` is what waits for an active
// one. Measured on Firefox 154: an install handler that rejects leaves
// `ready` unsettled past ten seconds while `register()` returns in 7 ms.
get ready() {
const healed = PLAN.activeAfterUnregister && log.unregisters > 0;
return (PLAN.readySettles || healed)
? Promise.resolve({}) : new Promise(() => {});
},
addEventListener: (type, fn) => { if (type === 'controllerchange') listeners.add(fn); },
removeEventListener: (type, fn) => { listeners.delete(fn); },
},
},
});
globalThis.window = globalThis;
globalThis.isSecureContext = true;
// The self-test's repair reloads once and remembers it for the tab; both have
// to exist here or priming the worker throws instead of repairing.
const session = new Map();
globalThis.sessionStorage = {
getItem: k => (session.has(k) ? session.get(k) : null),
setItem: (k, v) => session.set(k, String(v)),
removeItem: k => session.delete(k),
};
log.reloads = 0;
globalThis.location = { reload: () => { log.reloads += 1; } };
globalThis.document = {
createElement: () => ({ hidden: false, src: '', remove() {} }),
body: {
appendChild: (frame) => {
log.navigations += 1;
const port = pendingByFrame.get(frame.src);
const answer = PLAN.serveOnNavigation === 'always'
|| (PLAN.serveOnNavigation === 'second' && log.navigations >= 2);
if (port && answer) {
log.served += 1;
setTimeout(() => {
port.postMessage({type: 'mbdl-serving', id: frame.src});
// The worker's own copy of the port, dropped once answered. sw.js
// drops it with the pending entry; here it has to be explicit or the
// harness process never exits.
port.close();
}, 0);
}
},
},
};
const M = await import('%(module)s');
// Production waits 15 s for each; these cases are about which branch runs.
const FAST = {controlMs: %(control)d, servedMs: 400};
const out = {};
"""
def _run(tmp_path, body, *, control_after_ms=0, active=True,
serve="always", register_throws=False, control_budget_ms=800,
ready_settles=True, register_hangs=False,
active_after_unregister=False, control_on_claim=False):
module = tmp_path / "downloads.mjs"
module.write_text(DOWNLOADS.read_text())
(tmp_path / "package.json").write_text('{"type":"module"}')
plan = {
"controlAfterMs": control_after_ms,
"active": active,
"serveOnNavigation": serve,
"registerThrows": register_throws,
"readySettles": ready_settles,
"registerHangs": register_hangs,
"activeAfterUnregister": active_after_unregister,
"controlOnClaim": control_on_claim,
}
script = tmp_path / "case.mjs"
script.write_text(
(PRELUDE % {"plan": json.dumps(plan), "module": module.as_posix(),
"control": control_budget_ms})
+ body
+ "\nout.log = log;\nconsole.log(JSON.stringify(out));\n")
proc = subprocess.run(["node", str(script)], capture_output=True, text=True,
timeout=120)
assert proc.returncode == 0, proc.stderr
return json.loads(proc.stdout)
# ── A failure must never be cached ──────────────────────────────────────────
def test_a_missed_claim_does_not_poison_the_page(tmp_path):
"""
The bug: `_swReady` held the null, so every later download in that tab got
it back without trying. One slow first click and the tab could not stream
again — on Firefox, that is every large download for the rest of the visit.
Here the worker never takes control, so the first call fails; the second
must register again rather than return a remembered null.
"""
r = _run(tmp_path, """
out.first = await M.openStreamedDownload('a.bin', 10, FAST) !== null;
const after = log.registers;
out.second = await M.openStreamedDownload('b.bin', 10, FAST) !== null;
out.registeredAgain = log.registers > after;
""", control_after_ms=None)
assert r["first"] is False and r["second"] is False
assert r["registeredAgain"] is True, "a failed attempt was cached"
def test_a_success_is_reused_rather_than_re_registered(tmp_path):
"""The other half: once controlled, it must not re-register per download."""
r = _run(tmp_path, """
// Closed, like a real caller: an open target holds a keep-alive interval
// for the worker, and a test that leaks one never lets Node exit.
for (const name of ['a.bin', 'b.bin']) {
const t = await M.openStreamedDownload(name, 10, FAST);
out[name[0]] = t !== null;
if (t) await t.writable.close();
}
""")
assert r["a"] and r["b"]
assert r["log"]["registers"] <= 1, "re-registered on a page already controlled"
# ── Waiting for control, rather than giving up ──────────────────────────────
def test_control_arriving_late_is_still_used(tmp_path):
"""
Control used to be waited for with a 3 s cap, inside the click. A cold
worker on a busy machine can take longer, and the old code called that a
browser that cannot stream. Scaled down here — the budget is a parameter, so
what is pinned is that a claim arriving after the first check is still used,
not the particular number of seconds.
"""
r = _run(tmp_path, """
const t0 = Date.now();
const target = await M.openStreamedDownload('film.mkv', 20e9, FAST);
out.ok = target !== null;
out.waitedMs = Date.now() - t0;
if (target) await target.writable.close();
""", control_after_ms=1200, control_budget_ms=6000)
assert r["ok"] is True, "gave up on a claim that arrived late"
assert r["waitedMs"] >= 1100, "did not actually wait for the claim"
def test_an_uncontrolled_page_asks_the_worker_to_claim_again(tmp_path):
"""
Active but not controlling — a page loaded before any worker existed, whose
claim was missed. Rather than declare the path unavailable, ask again.
"""
r = _run(tmp_path, """
out.ok = await M.openStreamedDownload('a.bin', 10, FAST) !== null;
""", control_after_ms=None, active=True)
assert r["log"]["claims"] >= 1, "never asked the active worker to claim"
# ── Retrying a missed navigation ────────────────────────────────────────────
def test_a_missed_navigation_is_retried(tmp_path):
"""
The worker takes the stream and is then never asked for the URL. The page
used to give up at once; on Firefox that sends a film to the in-memory
floor. It gets a second go, with a fresh id and a fresh iframe.
"""
r = _run(tmp_path, """
const t = await M.openStreamedDownload('film.mkv', 20e9, FAST);
out.ok = t !== null;
if (t) await t.writable.close();
""", serve="second")
assert r["ok"] is True, "one missed navigation ended the download"
assert r["log"]["navigations"] == 2
def test_giving_up_says_why(tmp_path):
"""
A silent null is what made the original defect invisible. Whatever happens,
the reason has to be readable afterwards — it is what the refusal quotes.
"""
r = _run(tmp_path, """
out.target = await M.openStreamedDownload('a.bin', 10, FAST);
out.why = M.lastStreamFailure();
""", control_after_ms=None)
assert r["target"] is None
assert r["why"], "declined with no stated reason"
def test_a_registration_that_throws_is_reported_not_swallowed(tmp_path):
r = _run(tmp_path, """
out.target = await M.openStreamedDownload('a.bin', 10, FAST);
out.why = M.lastStreamFailure();
""", register_throws=True)
assert r["target"] is None
assert "registration" in r["why"]
# ── Wiring that the behavioural cases cannot see ────────────────────────────
def test_the_worker_is_primed_at_boot_not_at_the_first_click(tmp_path):
"""
Registration inside the first download is the whole reason the claim was
ever raced. `primeServiceWorker` has to be called where the app starts, and
from a module that actually imports it — `node --check` would not notice a
missing import, which is a mistake this repo has already shipped once.
"""
app = (STATIC / "app.js").read_text()
assert "downloads.primeServiceWorker()" in app, "nothing primes the worker"
assert "import * as downloads from './downloads.js'" in app, (
"app.js calls downloads.primeServiceWorker() without importing downloads")
# In mount(), which runs at start-up — not inside a component or a handler.
mount = app[app.index("const mount = () => {"):]
assert "downloads.primeServiceWorker()" in mount[:mount.index("\n};")]
def test_the_worker_answers_a_re_claim(tmp_path):
"""The page's last resort before declaring the path unavailable only works
if sw.js implements the other half."""
sw = (STATIC / "sw.js").read_text()
assert "mbdl-claim" in sw and "clients.claim()" in sw
# ── Nothing on this path may wait for ever ──────────────────────────────────
def test_a_worker_that_never_installs_does_not_hang_every_download(tmp_path):
"""The one that reached a person: four downloads stuck at "preparing", for
ever, with nothing in the node's journal because no transfer had been asked
for yet.
`register()` resolves as soon as the registration object exists — with
nothing but an *installing* worker — and `ready` waits for an active one.
Measured on Firefox 154: an install handler that rejects leaves `ready`
unsettled past ten seconds while `register()` returns in seven
milliseconds. Neither had a deadline, and `_swPromise` is shared, so every
download on the page waited on the same promise that would never settle.
"""
out = _run(tmp_path, """
const t0 = Date.now();
out.worker = await M.openStreamedDownload('film.mkv', 1, FAST);
out.ms = Date.now() - t0;
out.why = M.lastStreamFailure();
""", ready_settles=False, active=False, control_after_ms=None,
control_budget_ms=300)
assert out["worker"] is None
assert out["ms"] < 8000, (
f"gave up after {out['ms']}ms — a budget that is not enforced is not a "
"budget, and the row above it says 'preparing' the whole time")
assert "active" in out["why"], out["why"]
def test_a_stuck_ready_does_not_throw_away_a_working_worker(tmp_path):
"""`ready` can be waiting on a *newer* worker that cannot install while an
older one is perfectly able to serve. Giving up then would cost Firefox the
only unbounded way it has to write a download to disk — a deadline must
bound the waiting, never remove the capability."""
out = _run(tmp_path, """
const target = await M.openStreamedDownload('film.mkv', 1, FAST);
out.target = target !== null;
// Closing stops the keep-alive; left open, its interval keeps this process
// alive well past the test's own timeout.
if (target) await target.writable.close();
""", ready_settles=False, active=True, control_budget_ms=300)
assert out["target"] is True
def test_a_registration_that_never_answers_gives_up_too(tmp_path):
"""The other unbounded await. It rejects loudly in the case above; this is
the case where it says nothing at all."""
out = _run(tmp_path, """
const t0 = Date.now();
out.worker = await M.openStreamedDownload('film.mkv', 1, FAST);
out.ms = Date.now() - t0;
out.why = M.lastStreamFailure();
""", register_hangs=True, active=False, control_after_ms=None,
control_budget_ms=300)
assert out["worker"] is None
assert out["ms"] < 8000, f"gave up after {out['ms']}ms"
assert "register" in out["why"], out["why"]
def test_a_registration_stuck_installing_is_discarded_and_asked_for_again(tmp_path):
"""A deadline turns an invisible hang into a named failure, which is better
but is not a fix: a registration stuck with nothing but an installing worker
does not heal on its own. Every later visit finds the same registration and
waits on the same `ready`, so the browser stays unable to stream a download
until somebody opens developer tools — and on Firefox there is nothing else
that can write a film to disk.
So the stuck registration is thrown away and asked for once more.
"""
out = _run(tmp_path, """
const target = await M.openStreamedDownload('film.mkv', 1, FAST);
out.target = target !== null;
if (target) await target.writable.close();
""", ready_settles=False, active=False, control_after_ms=None,
active_after_unregister=True, control_budget_ms=300)
assert out["log"]["unregisters"] == 1, (
"the stuck registration was left in place")
assert out["target"] is True, (
"discarding it did not get the page a worker it could stream to")
# ── A page the worker cannot serve ──────────────────────────────────────────
def test_a_page_the_worker_cannot_serve_reloads_itself_once(tmp_path):
"""Being controlled is not being servable, and the gap is a real failure.
A document fetched by a hard reload — Ctrl+F5, Ctrl+Shift+R — is loaded with
the service worker bypassed. It can be claimed afterwards, so `controller`
comes back and every check in `_claimController` passes; but the navigations
it starts keep missing the worker, and the hidden iframe a streamed download
needs is a navigation. Every download then fails with "the worker did not
answer" for the life of that page — on Firefox and Safari, the only path
there is for a file too large to hold in memory.
Reported after an operator was told to hard-reload after each deployment:
four downloads out of four worked on a freshly started browser, and the
first attempt after a Ctrl+F5 failed, every time. An ordinary reload puts
the document back under the worker, so priming does exactly that, once.
"""
out = _run(tmp_path, """
M.primeServiceWorker();
await new Promise((r) => setTimeout(r, 6000));
out.reloads = log.reloads;
""", serve="never")
assert out["reloads"] == 1, (
"a page that cannot be served by the worker was left that way")
def test_a_page_that_works_is_not_reloaded(tmp_path):
"""The self-test costs milliseconds when it passes, and must cost nothing
else. Reloading a healthy page at boot would be a flicker on every visit."""
out = _run(tmp_path, """
M.primeServiceWorker();
await new Promise((r) => setTimeout(r, 3000));
out.reloads = log.reloads;
""")
assert out["reloads"] == 0
def test_the_repair_happens_at_most_once(tmp_path):
"""The flag is in sessionStorage rather than a variable because the point is
to survive the reload it triggers. If reloading does not help, the page
stays broken and says so — it does not reload again, and again."""
out = _run(tmp_path, """
sessionStorage.setItem('meshbay.sw-repaired', '1');
M.primeServiceWorker();
await new Promise((r) => setTimeout(r, 6000));
out.reloads = log.reloads;
""", serve="never")
assert out["reloads"] == 0, "a page that had already been repaired reloaded again"
def test_the_self_test_leaves_no_file_behind(tmp_path):
"""It opens a real download target to ask a real question, so it must also
tear it down: a completed one would drop `meshbay-selftest.bin` into the
download folder on every page load."""
src = DOWNLOADS.read_text()
fn = src[src.index("async function _canServeDownloads"):]
fn = fn[:fn.index("\n}\n")]
assert "writable.abort" in fn, (
"the self-test's stream is never aborted, so the browser keeps what it "
"was given")
assert "frame.remove" in fn
# ── The claim is asked for, not waited for ──────────────────────────────────
def test_an_uncontrolled_page_asks_at_once_rather_than_after_the_budget(tmp_path):
"""A page that is uncontrolled while an active worker exists will not be
claimed on its own — a document fetched by a hard reload is exactly that
shape. Waiting the whole control budget first spends it on something that
is not coming: about thirty seconds, measured, during which the person
clicks download and watches four rows hang before the page repairs itself.
"""
out = _run(tmp_path, """
const t0 = Date.now();
const target = await M.openStreamedDownload('film.mkv', 20e9, FAST);
out.ms = Date.now() - t0;
out.target = target !== null;
out.claims = log.claims;
// Closing stops the keep-alive; left open, its interval outlives the test.
if (target) await target.writable.close();
""", control_after_ms=None, control_on_claim=True, control_budget_ms=6000)
assert out["target"] is True
assert out["claims"] >= 1
assert out["ms"] < 3000, (
f"took {out['ms']}ms of a 6000ms budget — the claim was asked for only "
"after the wait, not before it")
def test_a_download_waits_for_the_self_test(tmp_path):
"""A click that lands while the check is still running must not race it.
On a page that turns out to be unservable it would otherwise spend the full
two attempts failing on a path that is about to be repaired."""
out = _run(tmp_path, """
M.primeServiceWorker();
const t0 = Date.now();
const target = await M.openStreamedDownload('film.mkv', 20e9, FAST);
out.ms = Date.now() - t0;
out.target = target !== null;
if (target) await target.writable.close();
""")
assert out["target"] is True
assert out["ms"] >= 1, "the download did not wait for priming at all"
|