diff options
| author | Christophe Besson <cbesson@gmail.com> | 2026-09-25 18:59:20 +0200 |
|---|---|---|
| committer | Christophe Besson <cbesson@gmail.com> | 2026-09-25 18:59:20 +0200 |
| commit | 2657ffd62ece8b8461d55b398139503ec504c3c6 (patch) | |
| tree | 0de6e0fadcdd1e989212613239239ee9d3b1a570 /packages/meshbay-node/tests | |
| parent | 3f3c67a4aff7b800c271e88e2bc5e5294b010fb9 (diff) | |
| download | meshbay-2657ffd62ece8b8461d55b398139503ec504c3c6.tar.gz | |
fix(node,client): survive a transient hub state on login, and surface a failed node-key link
Two defensive gaps turned a routine reset-and-reonboard into "impossible de
démarrer le node":
1. daemon._login_with_retry retried a 401 (node key not linked yet) but `raise`d
on every other status, so a 429 — the daemon's own 5s retries hitting the
sign-in rate limit — or a 502/503 while the hub restarts during a deploy
killed the process, and systemd crash-looped it. Those statuses (429, 5xx)
are now retried with a back-off that respects Retry-After, so a freshly
reset node stays alive (the operator needs it up to read its key) instead of
dying. A genuine 4xx (400/422) still raises.
2. create-group's linkNodeKey swallowed every error as "already linked or same
key" — but PUT /me/node_key is idempotent and returns 200 on a re-link, so
there was no benign error to hide: the catch only ever hid a real failure
(a rejected session, a bad key), letting the wizard proceed against a node
that looked linked but was not, which then could not authenticate. The link
failure now surfaces (detectNode shows it).
test_login_retry_is_resilient.py holds the retry behaviour (429/5xx retried,
Retry-After honoured, 401 stays alive, 400 still raises); red before, green
after. common/node/hub suites green.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat (limited to 'packages/meshbay-node/tests')
| -rw-r--r-- | packages/meshbay-node/tests/test_login_retry_is_resilient.py | 109 |
1 files changed, 109 insertions, 0 deletions
diff --git a/packages/meshbay-node/tests/test_login_retry_is_resilient.py b/packages/meshbay-node/tests/test_login_retry_is_resilient.py new file mode 100644 index 0000000..e37a413 --- /dev/null +++ b/packages/meshbay-node/tests/test_login_retry_is_resilient.py @@ -0,0 +1,109 @@ +"""A transient hub state on node sign-in must not crash the daemon. + +`_login_with_retry` retries a 401 (the node key is not linked yet — a human has +to link it, and the daemon must stay alive so its key can be read). It used to +`raise` on every other status, so a **429** (the daemon's own retries hitting +the sign-in rate limit) or a **502/503** (the hub restarting during a deploy) +killed the process — systemd then crash-looped it, which is what "impossible de +démarrer le node" looked like after a reset left the node with a fresh, unlinked +key. Those transient statuses are now retried with a back-off that respects +`Retry-After`. +""" + +import asyncio + +import httpx +import pytest +from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey +from meshbay_node.config import Config, GroupConfig, HubConfig, KeystoreConfig, NodeConfig +from meshbay_node.daemon import NodeDaemon + + +def _daemon(tmp_path): + cfg = Config( + hub=HubConfig(url="http://localhost:9999", username="testuser"), + node=NodeConfig(quic_port=29011, ui_port=29012), + groups=[], + keystore=KeystoreConfig(path=tmp_path / "keystore.enc"), + data_dir=tmp_path / "data", + ) + return NodeDaemon(cfg) + + +def _http_error(status: int, headers: dict | None = None) -> httpx.HTTPStatusError: + req = httpx.Request("POST", "http://localhost:9999/v1/nodes/auth") + resp = httpx.Response(status, headers=headers or {}, request=req) + return httpx.HTTPStatusError(f"{status}", request=req, response=resp) + + +class _Hub: + """A hub whose `startup` raises the given sequence, then returns a session.""" + def __init__(self, seq): + self._seq = list(seq) + self.calls = 0 + + async def startup(self, endpoint_hint=None): + self.calls += 1 + item = self._seq.pop(0) + if isinstance(item, Exception): + raise item + return item + + +@pytest.mark.asyncio +@pytest.mark.parametrize("status", [429, 500, 502, 503, 504]) +async def test_a_transient_status_is_retried_not_fatal(tmp_path, status, monkeypatch): + slept = [] + + async def _sleep(d): + slept.append(d) + monkeypatch.setattr(asyncio, "sleep", _sleep) + + daemon = _daemon(tmp_path) + session = object() + hub = _Hub([_http_error(status), session]) # transient, then success + got = await daemon._login_with_retry(hub) + assert got is session # it recovered instead of crashing + assert hub.calls == 2 # retried once + assert slept # it backed off + + +@pytest.mark.asyncio +async def test_retry_after_is_respected(tmp_path, monkeypatch): + slept = [] + + async def _sleep(d): + slept.append(d) + monkeypatch.setattr(asyncio, "sleep", _sleep) + + daemon = _daemon(tmp_path) + hub = _Hub([_http_error(429, {"Retry-After": "42"}), object()]) + await daemon._login_with_retry(hub) + assert 42 in slept + + +@pytest.mark.asyncio +async def test_a_401_still_retries_and_stays_alive(tmp_path, monkeypatch): + async def _sleep(d): + pass + monkeypatch.setattr(asyncio, "sleep", _sleep) + + daemon = _daemon(tmp_path) + session = object() + hub = _Hub([_http_error(401), session]) + got = await daemon._login_with_retry(hub) + assert got is session + assert daemon._state.get("status") in ("waiting_for_node_key", "waiting_for_account") + + +@pytest.mark.asyncio +async def test_a_genuine_client_error_still_raises(tmp_path, monkeypatch): + """A 400/422 is a bug, not a transient state — it must not be swallowed.""" + async def _sleep(d): + pass + monkeypatch.setattr(asyncio, "sleep", _sleep) + + daemon = _daemon(tmp_path) + hub = _Hub([_http_error(400), object()]) + with pytest.raises(httpx.HTTPStatusError): + await daemon._login_with_retry(hub) |