""" The service-worker download path, which on Firefox and Safari is the only unbounded way to write a file to disk. Neither of those browsers has the File System Access API, and OPFS is not a substitute: measured on Firefox 154, its quota is exactly 10% of the volume's size (389,233,459 bytes on a 3,892,334,592-byte volume, refused to the byte), which a film exceeds. So when this path declines, a large download has nowhere left to go — there is no floor under it that can hold a film. That is what makes its reliability a correctness property rather than a nicety. The real module is imported under Node with the browser pieces it reaches stubbed — `navigator.serviceWorker`, a document that "navigates" an iframe, and Node's own TransformStream and MessageChannel, which are the real ones. What is modelled is the environment; `serviceWorker()` and `openStreamedDownload()` are executed, never reimplemented. Three failures are pinned, all of which shipped: - registration happened inside the first click, so that click paid install, activate and claim while somebody watched a button do nothing; - a null result was cached for the life of the page, so one slow first click left the tab unable to stream anything again, curable only by a reload nobody knew to do; - one missed navigation fell straight through instead of retrying. """ import json import shutil import subprocess from pathlib import Path import pytest STATIC = Path(__file__).resolve().parents[1] / "src" / "meshbay_hub" / "static" DOWNLOADS = STATIC / "downloads.js" pytestmark = pytest.mark.skipif( shutil.which("node") is None or not DOWNLOADS.exists(), reason="node or the SPA sources are not available") # The stub browser. `plan` decides how the fake worker behaves, so one harness # covers every case below. PRELUDE = """ const store = new Map(); globalThis.localStorage = { getItem: k => (store.has(k) ? store.get(k) : null), setItem: (k, v) => store.set(k, String(v)), removeItem: k => store.delete(k), }; const PLAN = %(plan)s; const log = { registers: 0, claims: 0, navigations: 0, served: 0, unregisters: 0 }; // The worker as the page sees it: something with postMessage. It answers a // navigation by posting mbdl-serving back on the port it was handed, which is // exactly the confirmation the real sw.js sends from its fetch handler. let controller = null; const pendingByFrame = new Map(); const makeController = () => ({ postMessage: (msg, transfer) => { if (msg.type === 'mbdl-claim') { log.claims += 1; return; } if (msg.type !== 'mbdl') return; pendingByFrame.set('/_mbdl/' + msg.id, msg.port); }, }); const listeners = new Set(); // `globalThis.navigator` is read-only from Node 22 -- assigning to it is the // mistake CLAUDE.md already records against test_locales.py. Define it. Object.defineProperty(globalThis, 'navigator', { configurable: true, value: { serviceWorker: { get controller() { return controller; }, register: async () => { log.registers += 1; if (PLAN.registerThrows) throw new Error('registration blocked'); // A registration that never answers at all. Distinct from one that // rejects: nothing is reported, nothing fails, the caller just waits. if (PLAN.registerHangs) await new Promise(() => {}); // A worker that only becomes installable once the stuck registration // has been thrown away -- the browser this was reported from. const healed = PLAN.activeAfterUnregister && log.unregisters > 0; if (PLAN.controlAfterMs !== null || healed) { setTimeout(() => { controller = makeController(); for (const fn of listeners) fn(); }, healed ? 0 : PLAN.controlAfterMs); } return { active: (PLAN.active || healed) ? makeController() : null, unregister: async () => { log.unregisters += 1; return true; }, }; }, // `register()` resolves as soon as the registration object exists, with // nothing but an installing worker; `ready` is what waits for an active // one. Measured on Firefox 154: an install handler that rejects leaves // `ready` unsettled past ten seconds while `register()` returns in 7 ms. get ready() { const healed = PLAN.activeAfterUnregister && log.unregisters > 0; return (PLAN.readySettles || healed) ? Promise.resolve({}) : new Promise(() => {}); }, addEventListener: (type, fn) => { if (type === 'controllerchange') listeners.add(fn); }, removeEventListener: (type, fn) => { listeners.delete(fn); }, }, }, }); globalThis.window = globalThis; globalThis.isSecureContext = true; // The self-test's repair reloads once and remembers it for the tab; both have // to exist here or priming the worker throws instead of repairing. const session = new Map(); globalThis.sessionStorage = { getItem: k => (session.has(k) ? session.get(k) : null), setItem: (k, v) => session.set(k, String(v)), removeItem: k => session.delete(k), }; log.reloads = 0; globalThis.location = { reload: () => { log.reloads += 1; } }; globalThis.document = { createElement: () => ({ hidden: false, src: '', remove() {} }), body: { appendChild: (frame) => { log.navigations += 1; const port = pendingByFrame.get(frame.src); const answer = PLAN.serveOnNavigation === 'always' || (PLAN.serveOnNavigation === 'second' && log.navigations >= 2); if (port && answer) { log.served += 1; setTimeout(() => { port.postMessage({type: 'mbdl-serving', id: frame.src}); // The worker's own copy of the port, dropped once answered. sw.js // drops it with the pending entry; here it has to be explicit or the // harness process never exits. port.close(); }, 0); } }, }, }; const M = await import('%(module)s'); // Production waits 15 s for each; these cases are about which branch runs. const FAST = {controlMs: %(control)d, servedMs: 400}; const out = {}; """ def _run(tmp_path, body, *, control_after_ms=0, active=True, serve="always", register_throws=False, control_budget_ms=800, ready_settles=True, register_hangs=False, active_after_unregister=False): module = tmp_path / "downloads.mjs" module.write_text(DOWNLOADS.read_text()) (tmp_path / "package.json").write_text('{"type":"module"}') plan = { "controlAfterMs": control_after_ms, "active": active, "serveOnNavigation": serve, "registerThrows": register_throws, "readySettles": ready_settles, "registerHangs": register_hangs, "activeAfterUnregister": active_after_unregister, } script = tmp_path / "case.mjs" script.write_text( (PRELUDE % {"plan": json.dumps(plan), "module": module.as_posix(), "control": control_budget_ms}) + body + "\nout.log = log;\nconsole.log(JSON.stringify(out));\n") proc = subprocess.run(["node", str(script)], capture_output=True, text=True, timeout=120) assert proc.returncode == 0, proc.stderr return json.loads(proc.stdout) # ── A failure must never be cached ────────────────────────────────────────── def test_a_missed_claim_does_not_poison_the_page(tmp_path): """ The bug: `_swReady` held the null, so every later download in that tab got it back without trying. One slow first click and the tab could not stream again — on Firefox, that is every large download for the rest of the visit. Here the worker never takes control, so the first call fails; the second must register again rather than return a remembered null. """ r = _run(tmp_path, """ out.first = await M.openStreamedDownload('a.bin', 10, FAST) !== null; const after = log.registers; out.second = await M.openStreamedDownload('b.bin', 10, FAST) !== null; out.registeredAgain = log.registers > after; """, control_after_ms=None) assert r["first"] is False and r["second"] is False assert r["registeredAgain"] is True, "a failed attempt was cached" def test_a_success_is_reused_rather_than_re_registered(tmp_path): """The other half: once controlled, it must not re-register per download.""" r = _run(tmp_path, """ // Closed, like a real caller: an open target holds a keep-alive interval // for the worker, and a test that leaks one never lets Node exit. for (const name of ['a.bin', 'b.bin']) { const t = await M.openStreamedDownload(name, 10, FAST); out[name[0]] = t !== null; if (t) await t.writable.close(); } """) assert r["a"] and r["b"] assert r["log"]["registers"] <= 1, "re-registered on a page already controlled" # ── Waiting for control, rather than giving up ────────────────────────────── def test_control_arriving_late_is_still_used(tmp_path): """ Control used to be waited for with a 3 s cap, inside the click. A cold worker on a busy machine can take longer, and the old code called that a browser that cannot stream. Scaled down here — the budget is a parameter, so what is pinned is that a claim arriving after the first check is still used, not the particular number of seconds. """ r = _run(tmp_path, """ const t0 = Date.now(); const target = await M.openStreamedDownload('film.mkv', 20e9, FAST); out.ok = target !== null; out.waitedMs = Date.now() - t0; if (target) await target.writable.close(); """, control_after_ms=1200, control_budget_ms=6000) assert r["ok"] is True, "gave up on a claim that arrived late" assert r["waitedMs"] >= 1100, "did not actually wait for the claim" def test_an_uncontrolled_page_asks_the_worker_to_claim_again(tmp_path): """ Active but not controlling — a page loaded before any worker existed, whose claim was missed. Rather than declare the path unavailable, ask again. """ r = _run(tmp_path, """ out.ok = await M.openStreamedDownload('a.bin', 10, FAST) !== null; """, control_after_ms=None, active=True) assert r["log"]["claims"] >= 1, "never asked the active worker to claim" # ── Retrying a missed navigation ──────────────────────────────────────────── def test_a_missed_navigation_is_retried(tmp_path): """ The worker takes the stream and is then never asked for the URL. The page used to give up at once; on Firefox that sends a film to the in-memory floor. It gets a second go, with a fresh id and a fresh iframe. """ r = _run(tmp_path, """ const t = await M.openStreamedDownload('film.mkv', 20e9, FAST); out.ok = t !== null; if (t) await t.writable.close(); """, serve="second") assert r["ok"] is True, "one missed navigation ended the download" assert r["log"]["navigations"] == 2 def test_giving_up_says_why(tmp_path): """ A silent null is what made the original defect invisible. Whatever happens, the reason has to be readable afterwards — it is what the refusal quotes. """ r = _run(tmp_path, """ out.target = await M.openStreamedDownload('a.bin', 10, FAST); out.why = M.lastStreamFailure(); """, control_after_ms=None) assert r["target"] is None assert r["why"], "declined with no stated reason" def test_a_registration_that_throws_is_reported_not_swallowed(tmp_path): r = _run(tmp_path, """ out.target = await M.openStreamedDownload('a.bin', 10, FAST); out.why = M.lastStreamFailure(); """, register_throws=True) assert r["target"] is None assert "registration" in r["why"] # ── Wiring that the behavioural cases cannot see ──────────────────────────── def test_the_worker_is_primed_at_boot_not_at_the_first_click(tmp_path): """ Registration inside the first download is the whole reason the claim was ever raced. `primeServiceWorker` has to be called where the app starts, and from a module that actually imports it — `node --check` would not notice a missing import, which is a mistake this repo has already shipped once. """ app = (STATIC / "app.js").read_text() assert "downloads.primeServiceWorker()" in app, "nothing primes the worker" assert "import * as downloads from './downloads.js'" in app, ( "app.js calls downloads.primeServiceWorker() without importing downloads") # In mount(), which runs at start-up — not inside a component or a handler. mount = app[app.index("const mount = () => {"):] assert "downloads.primeServiceWorker()" in mount[:mount.index("\n};")] def test_the_worker_answers_a_re_claim(tmp_path): """The page's last resort before declaring the path unavailable only works if sw.js implements the other half.""" sw = (STATIC / "sw.js").read_text() assert "mbdl-claim" in sw and "clients.claim()" in sw # ── Nothing on this path may wait for ever ────────────────────────────────── def test_a_worker_that_never_installs_does_not_hang_every_download(tmp_path): """The one that reached a person: four downloads stuck at "preparing", for ever, with nothing in the node's journal because no transfer had been asked for yet. `register()` resolves as soon as the registration object exists — with nothing but an *installing* worker — and `ready` waits for an active one. Measured on Firefox 154: an install handler that rejects leaves `ready` unsettled past ten seconds while `register()` returns in seven milliseconds. Neither had a deadline, and `_swPromise` is shared, so every download on the page waited on the same promise that would never settle. """ out = _run(tmp_path, """ const t0 = Date.now(); out.worker = await M.openStreamedDownload('film.mkv', 1, FAST); out.ms = Date.now() - t0; out.why = M.lastStreamFailure(); """, ready_settles=False, active=False, control_after_ms=None, control_budget_ms=300) assert out["worker"] is None assert out["ms"] < 8000, ( f"gave up after {out['ms']}ms — a budget that is not enforced is not a " "budget, and the row above it says 'preparing' the whole time") assert "active" in out["why"], out["why"] def test_a_stuck_ready_does_not_throw_away_a_working_worker(tmp_path): """`ready` can be waiting on a *newer* worker that cannot install while an older one is perfectly able to serve. Giving up then would cost Firefox the only unbounded way it has to write a download to disk — a deadline must bound the waiting, never remove the capability.""" out = _run(tmp_path, """ const target = await M.openStreamedDownload('film.mkv', 1, FAST); out.target = target !== null; // Closing stops the keep-alive; left open, its interval keeps this process // alive well past the test's own timeout. if (target) await target.writable.close(); """, ready_settles=False, active=True, control_budget_ms=300) assert out["target"] is True def test_a_registration_that_never_answers_gives_up_too(tmp_path): """The other unbounded await. It rejects loudly in the case above; this is the case where it says nothing at all.""" out = _run(tmp_path, """ const t0 = Date.now(); out.worker = await M.openStreamedDownload('film.mkv', 1, FAST); out.ms = Date.now() - t0; out.why = M.lastStreamFailure(); """, register_hangs=True, active=False, control_after_ms=None, control_budget_ms=300) assert out["worker"] is None assert out["ms"] < 8000, f"gave up after {out['ms']}ms" assert "register" in out["why"], out["why"] def test_a_registration_stuck_installing_is_discarded_and_asked_for_again(tmp_path): """A deadline turns an invisible hang into a named failure, which is better but is not a fix: a registration stuck with nothing but an installing worker does not heal on its own. Every later visit finds the same registration and waits on the same `ready`, so the browser stays unable to stream a download until somebody opens developer tools — and on Firefox there is nothing else that can write a film to disk. So the stuck registration is thrown away and asked for once more. """ out = _run(tmp_path, """ const target = await M.openStreamedDownload('film.mkv', 1, FAST); out.target = target !== null; if (target) await target.writable.close(); """, ready_settles=False, active=False, control_after_ms=None, active_after_unregister=True, control_budget_ms=300) assert out["log"]["unregisters"] == 1, ( "the stuck registration was left in place") assert out["target"] is True, ( "discarding it did not get the page a worker it could stream to") # ── A page the worker cannot serve ────────────────────────────────────────── def test_a_page_the_worker_cannot_serve_reloads_itself_once(tmp_path): """Being controlled is not being servable, and the gap is a real failure. A document fetched by a hard reload — Ctrl+F5, Ctrl+Shift+R — is loaded with the service worker bypassed. It can be claimed afterwards, so `controller` comes back and every check in `_claimController` passes; but the navigations it starts keep missing the worker, and the hidden iframe a streamed download needs is a navigation. Every download then fails with "the worker did not answer" for the life of that page — on Firefox and Safari, the only path there is for a file too large to hold in memory. Reported after an operator was told to hard-reload after each deployment: four downloads out of four worked on a freshly started browser, and the first attempt after a Ctrl+F5 failed, every time. An ordinary reload puts the document back under the worker, so priming does exactly that, once. """ out = _run(tmp_path, """ M.primeServiceWorker(); await new Promise((r) => setTimeout(r, 6000)); out.reloads = log.reloads; """, serve="never") assert out["reloads"] == 1, ( "a page that cannot be served by the worker was left that way") def test_a_page_that_works_is_not_reloaded(tmp_path): """The self-test costs milliseconds when it passes, and must cost nothing else. Reloading a healthy page at boot would be a flicker on every visit.""" out = _run(tmp_path, """ M.primeServiceWorker(); await new Promise((r) => setTimeout(r, 3000)); out.reloads = log.reloads; """) assert out["reloads"] == 0 def test_the_repair_happens_at_most_once(tmp_path): """The flag is in sessionStorage rather than a variable because the point is to survive the reload it triggers. If reloading does not help, the page stays broken and says so — it does not reload again, and again.""" out = _run(tmp_path, """ sessionStorage.setItem('meshbay.sw-repaired', '1'); M.primeServiceWorker(); await new Promise((r) => setTimeout(r, 6000)); out.reloads = log.reloads; """, serve="never") assert out["reloads"] == 0, "a page that had already been repaired reloaded again" def test_the_self_test_leaves_no_file_behind(tmp_path): """It opens a real download target to ask a real question, so it must also tear it down: a completed one would drop `meshbay-selftest.bin` into the download folder on every page load.""" src = DOWNLOADS.read_text() fn = src[src.index("async function _canServeDownloads"):] fn = fn[:fn.index("\n}\n")] assert "writable.abort" in fn, ( "the self-test's stream is never aborted, so the browser keeps what it " "was given") assert "frame.remove" in fn