From 3c44f55f6b0aba77c7ad57d0a5ebe3e55b409473 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Tue, 25 Aug 2026 18:18:48 +0200 Subject: fix(hub,node): Create Group wizard silently skipped apps, and lost track of scanning progress MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two real-world bugs found together while testing multi-root group creation: - CreateGroupWizard only sent the enabled-apps PUT when the operator had *unchecked* something, assuming "every box left checked" already matched the node's own default (Roster.DEFAULT_APPS = chat, files). It doesn't — so leaving every app checked, the common case, silently left Videos/Music/Photos disabled on the node. Now sent unconditionally. - The wizard's "add extra roots" step never polled index-status, so once step 3 (which only watches the first/upload root) finished, the progress bar froze while the node kept scanning the remaining roots for minutes, unwatched. Added waitForRootsIndexed (platform.js), mirroring waitForGroupHosted's own race handling. That fix exposed a deeper one: indexer.py's _scan_root() only flipped `progress.scanning` on *after* walking the directory and stat()-ing every file — both off-loop, but slow enough on a large root that a poller's grace period (waitForRootsIndexed's 5s) could expire before ever observing `scanning: true` (confirmed against production logs: a GEK-init step fired 5.058s after a root started scanning, matching the grace period almost exactly). The stat() pass was also a synchronous loop directly on the asyncio event loop — blocking the whole daemon (WebRTC, chat, admin UI) for as long as it took on a root with many files. Both fixed: `scanning` now flips on before the walk starts, and stat()-ing is now off-loop too (_size_files). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_013XSohfUQQiaE77qyFLgSv3 --- packages/meshbay-node/tests/test_indexer.py | 85 +++++++++++++++++++++++++++++ 1 file changed, 85 insertions(+) (limited to 'packages/meshbay-node/tests/test_indexer.py') diff --git a/packages/meshbay-node/tests/test_indexer.py b/packages/meshbay-node/tests/test_indexer.py index 4bae620..9aa9fb9 100644 --- a/packages/meshbay-node/tests/test_indexer.py +++ b/packages/meshbay-node/tests/test_indexer.py @@ -2,6 +2,7 @@ import asyncio import os +import threading import time import pytest from pathlib import Path @@ -520,6 +521,90 @@ async def test_walk_root_does_not_stall_the_event_loop(tmp_path, sk_node, gek): "during a 0.2s walk") +@pytest.mark.asyncio +async def test_scanning_flag_is_true_while_the_walk_is_still_running(tmp_path, sk_node, gek): + """ + Regression for the actual bug this was found by (2026-08-25): `scanning` + used to flip on only *after* the walk finished, so a consumer polling + IndexProgress — index-status; the Create Group wizard's own progress + poll, which gives up after a short grace period if it never observes + `scanning: true` — read "not scanning" for however long a large/slow + root's discovery phase took, even though the node was already doing real + work. Confirmed against production logs: a wizard step waiting on this + flag gave up exactly at its grace-period deadline for a root whose walk + was still running. + """ + d = tmp_path / "shared" + d.mkdir() + (d / "f.bin").write_bytes(b"x") + + real_walk = indexer_mod._walk_root + walk_started = threading.Event() + release_walk = threading.Event() + + def slow_walk(root): + walk_started.set() + release_walk.wait(timeout=5) + return real_walk(root) + + indexer_mod._walk_root = slow_walk + indexer = DirectoryIndexer(roots=one_root(d), group_id="g", sk_node=sk_node, gek=gek) + try: + scan_task = asyncio.create_task(indexer.initial_scan()) + # Off-loop wait for the walk to actually start — busy-polling the + # event loop itself here would defeat the point of the test. + await asyncio.get_event_loop().run_in_executor(None, walk_started.wait, 5) + + assert indexer.progress.scanning is True, ( + "scanning must be True the moment the walk starts, not only " + "once it (and the sizing pass) finish") + finally: + release_walk.set() + indexer_mod._walk_root = real_walk + await scan_task + + assert indexer.progress.scanning is False + + +@pytest.mark.asyncio +async def test_scanning_flag_is_true_while_sizing_files_is_still_running( + tmp_path, sk_node, gek): + """ + Same regression, one phase later: sizing (stat()-ing every walked file) + used to be a synchronous loop straight on the asyncio event loop thread — + for a root with many thousands of files (a real personal library, not a + hypothetical) that blocked the entire daemon, and did so before + `scanning` was ever set. Now off-loop (_size_files) and `scanning` is + already true throughout, same principle as the walk phase above. + """ + d = tmp_path / "shared" + d.mkdir() + (d / "f.bin").write_bytes(b"x") + + real_size = indexer_mod._size_files + sizing_started = threading.Event() + release_sizing = threading.Event() + + def slow_size(files): + sizing_started.set() + release_sizing.wait(timeout=5) + return real_size(files) + + indexer_mod._size_files = slow_size + indexer = DirectoryIndexer(roots=one_root(d), group_id="g", sk_node=sk_node, gek=gek) + try: + scan_task = asyncio.create_task(indexer.initial_scan()) + await asyncio.get_event_loop().run_in_executor(None, sizing_started.wait, 5) + + assert indexer.progress.scanning is True + finally: + release_sizing.set() + indexer_mod._size_files = real_size + await scan_task + + assert indexer.progress.scanning is False + + @pytest.mark.asyncio async def test_reconcile_backoff_grows_with_no_changes_then_caps(shared_dir, sk_node, gek): indexer = DirectoryIndexer(roots=one_root(shared_dir), group_id="g", -- cgit v1.2.3