From d1f998b42137465b610667439527917a00030b4d Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Mon, 7 Sep 2026 16:02:10 +0200 Subject: fix(mnp): give a reply an id, so it stops being routed by luck MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit MNP carried no correlation id. A reply named its own type and nothing else, so a client with more than one request in flight worked out which one a message answered from the message itself — and for the replies that name nothing it could not. `_dispatch` fell through to matching by arrival order, which is a guess. `_sendAndWait` had the right value all along: it keys `_pending` by `this._seqId++` and never put it on the wire. The guess fails asymmetrically, which is why it hid. The victim is not the request that was answered wrongly — it is the unrelated one that now waits out its own 30s timeout for a reply already delivered elsewhere. Live on 2026-09-06: five `music_meta_req` sat pending for over 100 seconds behind a failing MusicBrainz, and a `device_list_result` was handed to one of them. The composer is disabled while a send is in flight, so a chat message whose reply went astray the same way left the Chat tab looking frozen for thirty seconds, then unfroze on its own. The `ack` half of this was fixed on 2026-08-30 by matching on request type. That closed the instance and left the class open: a refusal has no type to match on either, and `_dispatch_message`'s catch-all answers every unforeseen failure with `{"type": "error", "detail": "Request failed"}` — 238 of this module's 240 error sends name nothing at all. `req_id` now rides on the request and comes back on the reply. On the node it is published for the whole handler in a ContextVar and stamped by `_send`: a parameter would have meant threading an argument through all 240 send sites, and asyncio copies the context into a task, so a handler that `_spawn`s its real work still answers under the right id. It is never stamped on a broadcast — those answer nothing, and the owner check in `_send` is what keeps a chat broadcast or an index push from reaching another peer looking like a reply. On the client, `_dispatch` resolves on `req_id` first and the arrival-order fallback is gone the moment a node proves it stamps (`_correlates`, armed by the handshake's own reply). The fallback stays for an MNP 1.0 node, unchanged and no wider: there it is the only thing there is, and removing it would leave device_list_result, join_result and the handshake replies reaching nobody. Two things fall out. `sendChat` refuses an `error` reply like every other request in the file — it returned it as success, which did not matter while a refusal reached the wrong caller anyway and would now show a rejected message as sent. And `_group_ctx` uses `.get`: a reload pops a removed group while sessions connected to it are open, and every request they had left raised KeyError into that same catch-all. Sealed index messages are the one exception to the fast path. They cannot be handed over until they are opened, which is asynchronous while `_dispatch` is not — resolving on the id alone gave `fetchIndex` the envelope and skipped `onIndexSync` entirely. Caught by extending `index_seal_probe.mjs` to stamp a reply the way a current node does, after the hub suite passed over it: the probe built its own frames and had never seen one. Tests, all failing before and passing after: `test_chat_send.py` drives the real ChatPanel over the real transport for both shapes of reply with an older request pending (3 of its 6 are new, and the 3 for `ack` pass either way, so it discriminates); `test_reply_correlation.py` pins the node's half — the refusals that name nothing else, the broadcast that must not be stamped, and a late reply from a spawned task answering under its own id rather than the most recent request's. Full suite: 1897 passed, same 11 pre-existing failures as before. QUIC keeps its own dispatch and is not stamped. It is disabled by default and no browser request reaches it, but the asymmetry is real. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Dn1xYx9uT69mCB6UDvyKAN --- .../meshbay-hub/tests/test_index_seal_client.py | 30 +++++++++++++++++++++- 1 file changed, 29 insertions(+), 1 deletion(-) (limited to 'packages/meshbay-hub/tests/test_index_seal_client.py') diff --git a/packages/meshbay-hub/tests/test_index_seal_client.py b/packages/meshbay-hub/tests/test_index_seal_client.py index ca2c7a2..e1135da 100644 --- a/packages/meshbay-hub/tests/test_index_seal_client.py +++ b/packages/meshbay-hub/tests/test_index_seal_client.py @@ -45,10 +45,15 @@ def _frame(msg: dict) -> str: return (struct.pack(">I", len(body)) + body).hex() -def _sync_frame(gek: bytes, *names: str, version: int = 3) -> str: +def _sync_frame(gek: bytes, *names: str, version: int = 3, + req_id: int | None = None) -> str: payload = {"version": version, "entries": [_entry(n) for n in names], "dirs": ["library"], "roots": [{"name": "library"}]} + # `req_id` is what a current node stamps on a *reply*; the push it sends a + # newly connected peer answers no request and carries none. Both shapes + # arrive here, and only one of them may resolve a waiting fetchIndex. return _frame({"type": "index_sync", "v": "1.0", "group_id": GROUP, + **({"req_id": req_id} if req_id is not None else {}), **seal(gek, PURPOSE_INDEX, "index_sync", GROUP, payload)}) @@ -141,3 +146,26 @@ def test_deltas_are_applied_in_arrival_order(): assert [e["additions"][0] for e in out["events"][1:]] == [ "added-0.mkv", "added-1.mkv", "added-2.mkv"] assert [e["base_version"] for e in out["events"][1:]] == [3, 4, 5] + + +def test_a_sealed_reply_is_opened_before_it_reaches_its_caller(): + """A stamped index_sync must not be short-circuited by its `req_id`. + + Every other reply a node stamps is resolved straight out of the pending + map, which is the whole point of the id. An index message cannot be: it is + sealed, opening it is asynchronous, and `_dispatch` is not. Handing it over + on the strength of the id alone gives `fetchIndex` the envelope — nonce and + ciphertext, no entries — and never calls `onIndexSync` at all. + + `req_id` is 0 here because it is the transport's first request, and a + falsy id is exactly the one a presence check gets wrong. + """ + out = _run([_sync_frame(GEK, "a-film.mkv", req_id=0)]) + + assert [e["event"] for e in out["events"]] == ["index_sync"], ( + "the consumer was never told about an index that arrived as a reply") + assert out["events"][0]["entries"] == ["a-film.mkv"] + assert out["events"][0]["hasCiphertext"] is False + assert out["fetchIndex"]["state"] == "resolved" + assert out["fetchIndex"]["entries"] == ["a-film.mkv"], ( + "the caller was handed the sealed envelope instead of the index") -- cgit v1.2.3