summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/daemon.py
blob: 6d808aed335a6599149a7c1b8184fb3f0c2672f9 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
"""
MeshBay Node daemon — main process.

Startup sequence:
  1. Load config (~/.config/meshbay/node.toml)
  2. Load or create keystore (Argon2id unlock)
  3. Connect to hub: register → login → announce node
  4. Fetch GEK bundle from hub (if group configured)
  5. Start directory indexer (watchdog)
  6. Create chat stores (one SQLite DB per group)
  7. Create WebRTC transport (browser + native clients via DataChannel)
  8. Start QUIC chunk server (LAN / port-forwarded / hub-less direct access)
  9. (Phase 11.5: the unauthenticated HTTP file API and the TCP+TLS server were removed)
  10. Start hub WebSocket (signaling, revocations, WebRTC offers)
  11. Start local control API on node.ui_port (loopback only, token-gated)
  12. Run until SIGINT/SIGTERM

Usage:
  meshbay-node                    # interactive password prompt
  meshbay-node --config /path     # custom config
  meshbay-node status             # node state + public key (works while stopped)
  meshbay-node gek-init           # initialise the group key (no browser needed)
  meshbay-node init               # write example config + create keystore
  meshbay-node --calibrate-argon2 # benchmark Argon2id, suggest parameters
"""

import asyncio
import base64
import logging
import os
import signal
import sys
import time
from dataclasses import asdict
from pathlib import Path

import uvicorn
from meshbay_common import MNP_VERSION
from meshbay_common.background import spawn
from meshbay_common.paths import fold
from meshbay_common.protocol import MNP

from meshbay_node import uploads as uploads_mod
from meshbay_node.audit import RETENTION_DAYS as AUDIT_RETENTION_DAYS
from meshbay_node.audit import AuditStore
from meshbay_node.bundle_store import BundleStore
from meshbay_node.chat.store import ChatStore
from meshbay_node.cli.dispatch import run, start
from meshbay_node.config import DEFAULT_CONFIG_PATH, Config, load_config
from meshbay_node.enrichment import EnrichmentMixin
from meshbay_node.hub_client import HubClient, HubConfig
from meshbay_node.indexer import DirectoryIndexer, GroupIndex, IndexCache
from meshbay_node.indexer.enrich import Enricher
from meshbay_node.indexer.enrich_audio import AudioEnricher
from meshbay_node.indexer.enrich_photo import PhotoEnricher
from meshbay_node.keystore import load_or_create_keystore
from meshbay_node.media_cache import MediaCache
from meshbay_node.musicbrainz import MusicBrainzClient
from meshbay_node.platform import chmod_private
from meshbay_node.roots import RootError, RootSet, off_disk
from meshbay_node.roster import Roster
from meshbay_node.tmdb import TmdbClient
from meshbay_node.transport import (
    QUIC_AVAILABLE,
    WEBRTC_AVAILABLE,
    Denylist,
)
from meshbay_node.transport.wire import index_delta_message, index_sync_message

if QUIC_AVAILABLE:
    from meshbay_node.transport import QuicChunkServer
if WEBRTC_AVAILABLE:
    from meshbay_node.transport import WebRTCTransport

log = logging.getLogger(__name__)


# ── Hub WS sender bridge ─────────────────────────────────────────────────────

class _WsSender:
    """Thin bridge so WebRTC context can call hub_ws.send() for chat_notify."""

    def __init__(self, hub_client: HubClient):
        self._hub = hub_client

    async def send(self, data: str) -> None:
        await self._hub.send_ws(data)


# ── Daemon ────────────────────────────────────────────────────────────────────

def _root_shape(roots) -> set[tuple]:
    """What has to match for a group's roots to count as unchanged on reload."""
    return {(r.name, str(r.path), r.writable, r.removable) for r in roots}


class NodeDaemon(EnrichmentMixin):
    def __init__(self, config: Config, config_path: Path = DEFAULT_CONFIG_PATH):
        self._config = config
        self._config_path = config_path
        self._state: dict = {
            "status":        "starting",
            "hub_url":       config.hub.url,
            "username":      config.hub.username,
            "groups":        [g.name for g in config.groups],
            "quic_port":     config.node.quic_port,
            "endpoint_hint": None,
            "indexes":       {},
            "indexers":      {},
        }
        self._quic_server = None
        self._webrtc = None
        # Persisted so a restart does not silently un-revoke everyone (H4)
        self._denylist = (
            Denylist(path=config.data_dir / "denylist.json") if Denylist else None)
        self._chat_stores: dict[str, ChatStore] = {}
        # One instance, shared by every group's DirectoryIndexer — see
        # indexer/cache.py's docstring for why this stopped being per-group.
        self._index_cache: IndexCache | None = None
        # Coalesces a burst of index changes (one per debounced watchdog
        # event) into a single broadcast — see _on_index_change. 0.5s is
        # short enough nobody notices the wait, long enough that dropping a
        # few hundred files into a watched folder produces one push instead
        # of one per file.
        self._broadcast_coalesce_secs = 0.5
        self._pending_broadcasts: dict[str, asyncio.TimerHandle] = {}
        # group_id -> (version, {id: entry}) as of the last thing actually
        # broadcast — the comparison point for the next delta.
        self._last_broadcast_snapshot: dict[str, tuple] = {}
        self._audit_store: AuditStore | None = None
        self._bundle_store: BundleStore | None = None
        self._media_cache: MediaCache | None = None
        self._enricher: Enricher | None = None
        self._tmdb_client: TmdbClient | None = None
        self._audio_enricher: AudioEnricher | None = None
        self._musicbrainz_client: MusicBrainzClient | None = None
        self._photo_enricher: PhotoEnricher | None = None
        # A file id attempted at most once per daemon run, success or
        # failure — a persistently unprobeable file (corrupt, still being
        # written) does not get re-queued on every coalesced broadcast. A
        # restart retries everything, matching the "disposable, rebuildable"
        # stance the rest of this cache takes (docs/MESHBAY_DESIGN.md §6.5).
        # Shared across the video and audio enrichment paths — content-
        # addressed ids never collide between the two. Keyed by
        # (group_id, entry.id), not entry.id alone: the id is a content
        # hash, so the same physical file shared into two different groups
        # (found live — overlapping test libraries across several demo
        # groups) produces the same id in both. A bare-id set marked the
        # second group's copy "already attempted" the moment the first
        # group's enrichment ran, even though nothing had ever populated
        # *that* group's own index — every file in the second group stayed
        # at duration 0 with no artist/album, permanently, since nothing
        # ever revisits an id already in this set.
        self._enriched_attempted: set[tuple[str, str]] = set()
        self._roster: Roster | None = None
        self._indexers: list[DirectoryIndexer] = []
        self._tasks: list[asyncio.Task] = []
        self._hub: HubClient | None = None
        self._reload_lock = asyncio.Lock()

    async def run(self) -> None:
        log.info("MeshBay Node starting up")

        # 1. Keystore
        keys = load_or_create_keystore(
            path=self._config.keystore.path,
            unlock_file=self._config.keystore.unlock_file,
        )
        log.info("Keys loaded: %s", keys.pk_ed25519_b64[:16])

        # 2. Start the local control API early (so the operator can read the
        # node key before hub login). It is JSON-only, loopback-only, and both
        # the CLI and the desktop client's Node page are its clients.
        self._state["pk_node_ed25519"] = keys.pk_ed25519_b64
        self._state["config"] = self._config
        # Where it came from, so `group add` appends to the file this process
        # actually read rather than guessing at the default.
        self._state["config_path"] = str(self._config_path)
        # Per-run token for the control API (11.5.3). Not a password: it keeps
        # other local processes and rebound browser pages out of an API that can
        # re-initialise group keys.
        ui_token = base64.urlsafe_b64encode(os.urandom(18)).decode().rstrip("=")
        self._state["ui_token"] = ui_token
        # Persisted so the CLI and the desktop client can read it — nobody
        # should ever copy a token out of a log or a terminal.
        self._config.data_dir.mkdir(parents=True, exist_ok=True)
        self._ui_token_file = self._config.data_dir / "ui-token"
        self._ui_token_file.write_text(ui_token, encoding="utf-8", newline="\n")
        chmod_private(self._ui_token_file)
        from meshbay_node.ui import create_ui_app
        ui_app = create_ui_app(self._state)
        ui_cfg = uvicorn.Config(
            ui_app,
            host="127.0.0.1",
            port=self._config.node.ui_port,
            log_level="warning",
        )
        ui_server = uvicorn.Server(ui_cfg)
        self._tasks.append(asyncio.create_task(ui_server.serve()))
        log.info("Control API on 127.0.0.1:%d", self._config.node.ui_port)

        self._tasks.append(asyncio.create_task(self._reap_partial_uploads()))

        # 3. Hub connection (Ed25519 auth — retries until node key is linked)
        hub_cfg = HubConfig(
            hub_url=self._config.hub.url,
            username=self._config.hub.username,
        )
        async with HubClient(hub_cfg, keys) as hub:
            self._hub = hub
            session = await self._login_with_retry(hub)
            self._state["endpoint_hint"] = session.node_id

            try:
                session.email = await hub.fetch_owner_email()
            except Exception as e:
                log.warning("Could not fetch owner email: %s", e)

            # 4. Bundle store (P2P GEK bundles)
            data_dir = self._config.data_dir
            data_dir.mkdir(parents=True, exist_ok=True)
            self._bundle_store = BundleStore(db_path=data_dir / "bundles.db")
            await self._bundle_store.open()
            log.info("Bundle store opened: %s", data_dir / "bundles.db")

            # 4a. Path->hash cache, node-wide — opened once, shared by every
            # group's DirectoryIndexer below (indexer/cache.py).
            self._index_cache = IndexCache(db_path=data_dir / "index_cache.db")
            await self._index_cache.open()
            self._state["index_cache"] = self._index_cache
            log.info("Index cache opened: %s", data_dir / "index_cache.db")

            # 4b. Roster — who this node recognises and which keys are theirs.
            # Node authority is established here, locally, and never learned from
            # the hub: a hub that could name the operator's key could install
            # itself as node administrator.
            self._roster = Roster(db_path=data_dir / "roster.db")
            await self._roster.open()
            await self._roster.purge_expired()

            # Apply any roster overrides to node config (panel-edited values
            # take precedence over node.toml defaults).
            from meshbay_node.config import node_settings_defaults
            nd = self._config.node
            effective = await self._roster.node_settings(
                node_settings_defaults(nd))
            for k, v in effective.items():
                setattr(nd, k, v)

            # X25519 key material for GEK unwrapping
            from cryptography.hazmat.primitives import serialization
            sk_x_raw = keys.sk_x25519.private_bytes(
                serialization.Encoding.Raw, serialization.PrivateFormat.Raw,
                serialization.NoEncryption())
            pk_x_raw = base64.b64decode(keys.pk_x25519_b64)

            # 4. Build per-group contexts
            groups_ctx: dict[str, dict] = {}
            for group_cfg in self._config.groups:
                if not group_cfg.id or not group_cfg.roots:
                    log.warning("Group %r has no id or no shared directory — "
                                "skipping", group_cfg.name)
                    continue

                try:
                    roots = await self._build_roots(group_cfg)
                except RootError as e:
                    # Configuration the operator has to fix; guessing would put
                    # a member's file on the wrong disk or index one twice.
                    log.error("Group %r: %s — skipping", group_cfg.name, e)
                    continue

                await off_disk(roots, roots.refresh_availability)
                if not any(r.available for r in roots):
                    # Not skipped for being empty: a group whose only drive is
                    # unplugged still exists, and its index is frozen rather
                    # than lost. But there is nothing to serve until it returns.
                    log.warning(
                        "Group %r: none of its %d root(s) are readable right now "
                        "(%s) — serving nothing until one returns",
                        group_cfg.name, len(roots),
                        ", ".join(str(r.path) for r in roots))

                gek = None
                gek = await self._load_gek(
                    group_cfg.id, session.user_id, sk_x_raw, pk_x_raw)
                if gek:
                    log.info("GEK loaded for group %s", group_cfg.id[:8])
                else:
                    log.info("No GEK yet for group %s — will accept first setup",
                             group_cfg.name)

                # Read once at load, like enabled_apps below —
                # kept current in place afterwards by set_scan_settings
                # (ops.py), which updates both this indexer object directly
                # and roster.db, so a restart picks up the same values.
                scan_settings = (
                    await self._roster.scan_settings(group_cfg.id)
                    if self._roster else {
                        "reconcile_interval_secs": DirectoryIndexer.DEFAULT_RECONCILE_SECS,
                        "debounce_secs": DirectoryIndexer.DEFAULT_DEBOUNCE_SECS,
                    })

                indexer = DirectoryIndexer(
                    roots=roots,
                    group_id=group_cfg.id,
                    sk_node=keys.sk_ed25519,
                    gek=gek,
                    on_change=self._on_index_change,
                    on_root_ejected=self._eject_persister(group_cfg.id),
                    cache=self._index_cache,
                    reconcile_secs=scan_settings["reconcile_interval_secs"],
                    debounce_secs=scan_settings["debounce_secs"],
                )
                await indexer.start(defer_scan=True)
                self._indexers.append(indexer)
                self._state["indexes"][group_cfg.id] = indexer.index
                self._state["indexers"][group_cfg.id] = indexer
                log.info("Group %s configured: %s (scan deferred)",
                         group_cfg.name,
                         ", ".join(f"{r.name}={r.path}" for r in roots))

                groups_ctx[group_cfg.id] = {
                    "gek": gek,
                    "roots": roots,
                    "index": indexer.index,
                    # Live reference, mutated in place by the indexer itself
                    # (see IndexProgress in indexer.py) — read, never copied,
                    # by the handshake ack and the periodic progress pusher.
                    "progress": indexer.progress,
                    # Bound method, called when a peer completes the
                    # handshake — resets reconcile's backoff (indexer.py
                    # _reconcile_loop) so the backstop is prompt again now
                    # that someone is actually looking.
                    "note_activity": indexer.note_activity,
                    # Bound method, called when an upload finishes. The entry
                    # it belongs to does not exist yet (see
                    # webrtc/upload_handlers.py _register_uploader), so the indexer keeps
                    # the record and stamps the entry when it creates it.
                    "record_upload": indexer.record_upload,
                    # Shown to the operator in Settings, and kept current in
                    # place by set_scan_settings (ops.py) — same reasoning as
                    # enabled_apps below.
                    "reconcile_interval_secs": scan_settings["reconcile_interval_secs"],
                    "debounce_secs": scan_settings["debounce_secs"],
                    "visibility": group_cfg.visibility,
                    # Admission policy comes from node.toml, never from the hub:
                    # a hub that could declare a group open would be handed its key.
                    "join_policy": group_cfg.join_policy,
                    # Read once at load, kept current in place by the signed
                    # operation that changes it — the upload handler is
                    # synchronous and a database round trip per chunk would be
                    # absurd. (Whether a member may upload is not here any
                    # more: it is `writable` on the root being written to,
                    # which the RootSet above already carries.)
                    "enabled_apps": await self._roster.enabled_apps(
                        group_cfg.id) if self._roster else list(Roster.DEFAULT_APPS),
                    # How many transfers one member may run at once here. Empty
                    # means the operator has not said, and the node's default
                    # applies — never "unlimited" (transfers.member_cap).
                    "transfer_limits": (
                        await self._roster.transfer_limits(group_cfg.id)
                        if self._roster else {}),
                    # Which folder(s) inside the shared roots each app works
                    # over. One shape for every app (roster.py's
                    # app_directories) — an empty list means nothing has been
                    # chosen, which every app reads as "show nothing yet",
                    # never "the whole group index".
                    **(await self._app_directories_ctx(group_cfg.id)),
                    # Whether the node unfurls links members post here.
                    "chat_link_preview": await self._roster.chat_link_preview(
                        group_cfg.id) if self._roster else True,
                    # Whether members' cross-group Search lists this group.
                    "search_listed": await self._roster.search_listed(
                        group_cfg.id) if self._roster else True,
                    # Which chat epoch key is current. Opened here if the
                    # group has none, because chat is always encrypted (MNP
                    # 2.0) and a group with no epoch is a group nobody can
                    # speak in — the node cannot wait for an operator to notice.
                    "chat_epoch": await self._ensure_chat_epoch(group_cfg.id),
                    # Whether TMDB lookups run for this group at all —
                    # per-group (2026-08-24, used to be node-wide), same
                    # "read once, kept current in place by the signed op"
                    # shape as the app directories above.
                    "tmdb_enabled": await self._roster.tmdb_enabled(
                        group_cfg.id) if self._roster else True,
                    # Music app equivalent of tmdb_enabled — per-group from
                    # the start (docs/MESHBAY_DESIGN.md §9.8).
                    "musicbrainz_enabled": await self._roster.musicbrainz_enabled(
                        group_cfg.id) if self._roster else True,
                }

            if not groups_ctx:
                log.warning("No groups configured yet — the control API and hub "
                            "connection stay up; attach a group to go live")

            # 5. Chat stores (one SQLite DB per group)
            for gid in groups_ctx:
                chat_db = data_dir / gid[:16] / "chat.db"
                store = ChatStore(db_path=chat_db)
                await store.open()
                self._chat_stores[gid] = store
                groups_ctx[gid]["chat_store"] = store
            log.info("Chat stores opened: %d groups", len(self._chat_stores))

            # 6. Audit store (legal compliance — IP + action logging)
            audit_db = data_dir / "audit.db"
            self._audit_store = AuditStore(db_path=audit_db)
            await self._audit_store.open()
            log.info("Audit store opened: %s", audit_db)
            self._tasks.append(asyncio.create_task(self._purge_audit_log()))

            # 6b. Media cache (Videos app — TMDB metadata + thumbnails).
            # Node-wide like audit.db, not per-group: a thumbnail is the same
            # bytes regardless of which group happens to share the file
            # (docs/MESHBAY_DESIGN.md §6.5, §9.7).
            media_cache_db = data_dir / "media_cache.db"
            self._media_cache = MediaCache(db_path=media_cache_db)
            await self._media_cache.open()
            self._enricher = Enricher(self._media_cache)
            self._tmdb_client = TmdbClient(roster=self._roster)
            # Token/language only — read once at load, kept current in place
            # by ops.set_tmdb_config (the signed op), exposed to every
            # group's handshake ack via `daemon_state` (already wired to
            # self._webrtc._ctx below) since these stay node-wide, one
            # shared credential/cache. Whether TMDB is used at all is now
            # per-group instead — see each group's own "tmdb_enabled" in
            # groups_ctx above.
            tmdb_token, tmdb_language = await self._roster.tmdb_config()
            self._state["tmdb_token_customized"] = bool(tmdb_token)
            self._state["tmdb_language"] = tmdb_language or ""

            # 6c. Music app (docs/MESHBAY_DESIGN.md §9.8) — same
            # media_cache.db, its own enricher (mutagen, not ffmpeg) and its
            # own MusicBrainz client.
            # The User-Agent contact is the owner's hub email, resolved at
            # login — no roster setting or env var needed any more.
            self._audio_enricher = AudioEnricher(self._media_cache)
            self._musicbrainz_client = MusicBrainzClient(owner_email=session.email)
            self._state["musicbrainz_contact_configured"] = bool(session.email)

            # 6d. Photos app (docs/MESHBAY_DESIGN.md §9.9) — same
            # media_cache.db, its own enricher (Pillow, not ffmpeg/mutagen).
            # No credential, no
            # third-party client to construct: EXIF is read locally.
            self._photo_enricher = PhotoEnricher(self._media_cache)
            self._state["media_cache"] = self._media_cache
            log.info("Media cache opened: %s", media_cache_db)

            # 5. Denylist
            denylist = self._denylist

            # 6. WebRTC transport (browser clients)
            #
            # The operator's answer about the GPU, before the first stream asks
            # the question. The probe itself is lazy — it costs a test encode,
            # and a node that never serves a video should never pay it.
            from meshbay_node import hwaccel
            hwaccel.set_enabled(self._config.node.hardware_video_encode)
            from meshbay_node.transport.ice_filter import install as install_ice_filter
            install_ice_filter(
                self._config.node.ice_interfaces or None,
            )
            # aiortc keeps only the FIRST entry of RTCConfiguration.iceServers, so
            # the multi-STUN fallback only exists on the node if aioice itself
            # fans out — see transport/stun_multi.
            from meshbay_node.transport.stun_multi import install as install_stun_multi
            install_stun_multi(self._config.node.stun_servers or None)

            first = next(iter(groups_ctx.values()), None)
            if WEBRTC_AVAILABLE:
                self._webrtc = WebRTCTransport(
                    sk_node=keys.sk_ed25519,
                    hub_pk_pem=session.hub_pk_pem,
                    gek=first["gek"] if first else None,
                    roots=first["roots"] if first else None,
                    index=first["index"] if first else None,
                    groups=groups_ctx,
                    denylist=denylist,
                    max_concurrent_streams=self._config.node.max_concurrent_streams,
                    max_concurrent_downloads=self._config.node.max_concurrent_downloads,
                    max_concurrent_uploads=self._config.node.max_concurrent_uploads,
                    max_upload_gb=self._config.node.max_upload_gb,
                    transcode_incompatible_video=self._config.node.transcode_incompatible_video,
                    stun_servers=self._config.node.stun_servers or None,
                )
                # No global chat_store here: each group's store lives in
                # groups_ctx[gid]["chat_store"] and is resolved per session via
                # _group_ctx(). Assigning the first group's store transport-wide
                # sent every group's chat to one database and served it back to
                # members of every other group (finding H1).
                self._webrtc._ctx["groups"] = groups_ctx
                self._webrtc._ctx["hub_ws"] = _WsSender(hub)
                self._webrtc._ctx["node_user_id"] = session.user_id
                self._webrtc._ctx["audit_store"] = self._audit_store
                self._webrtc._ctx["bundle_store"] = self._bundle_store
                self._webrtc._ctx["media_cache"] = self._media_cache
                self._webrtc._ctx["tmdb_client"] = self._tmdb_client
                self._webrtc._ctx["musicbrainz_client"] = self._musicbrainz_client
                self._webrtc._ctx["sk_x25519_raw"] = sk_x_raw
                self._webrtc._ctx["pk_x25519_raw"] = pk_x_raw
                self._webrtc._ctx["pk_x25519_b64"] = keys.pk_x25519_b64

                self._webrtc._ctx["roster"] = self._roster
                # The MNP adapter calls the same operations as the loopback API
                # (meshbay_node.ops), and those take the daemon's state. Handing
                # the transport a second set of lookups is how two paths to one
                # operation start disagreeing — the shape of C1 and C6.
                self._webrtc._ctx["daemon_state"] = self._state
                self._webrtc._ctx["invite_ttl"] = (
                    self._config.node.invite_ttl_hours * 3600)
                self._webrtc._ctx["device_request_ttl"] = (
                    self._config.node.device_request_ttl_minutes * 60)
                paired = await self._roster.has_operator() if self._roster else False
                self._webrtc._ctx["has_admin_authority"] = paired
                if paired:
                    log.info("Node authority: paired operator")
                else:
                    log.warning(
                        "No operator paired — invites and file deletion are "
                        "refused. Run: meshbay-node operator pair")
                log.info("WebRTC transport ready")
            else:
                log.warning("WebRTC not available (aiortc not installed)")

            # 7. QUIC chunk server (LAN / port-forwarded / hub-less direct access)
            #
            # Off unless `[node] quic_enabled = true`: no shipping client speaks
            # QUIC (browser and desktop use WebRTC; the `group://` sidecar is
            # unbuilt), so starting it by default only exposes a UDP port.
            if QUIC_AVAILABLE and self._config.node.quic_enabled:
                self._quic_server = QuicChunkServer(
                    sk_node=keys.sk_ed25519,
                    hub_pk_pem=session.hub_pk_pem,
                    gek=first["gek"] if first else None,
                    roots=first["roots"] if first else None,
                    index=first["index"] if first else None,
                    host="::",
                    port=self._config.node.quic_port,
                    groups=groups_ctx,
                    denylist=denylist,
                )
                self._quic_server._ctx["groups"] = groups_ctx
                await self._quic_server.start()
                log.info("QUIC server on port %d (%d groups)",
                         self._config.node.quic_port, len(groups_ctx))
            elif QUIC_AVAILABLE:
                log.info("QUIC server disabled ([node] quic_enabled = false)")

            # 8. Hub WebSocket (signaling + revocations + WebRTC offers)
            async def on_webrtc_offer(sdp, peer_id, ice_candidates):
                if not self._webrtc:
                    return None
                try:
                    answer_sdp, answer_ice = await self._webrtc.handle_offer(
                        sdp, peer_id)
                    log.info("WebRTC answer for peer=%s (%d peers)",
                             peer_id, self._webrtc.active_peers)
                    return (answer_sdp, answer_ice)
                except Exception as e:
                    log.error("WebRTC offer failed: %s", e)
                    return None

            async def on_incoming(peer_ip, peer_port):
                if self._quic_server:
                    self._quic_server.punch_nat(peer_ip, peer_port)

            def on_revocation(token):
                if denylist and token:
                    import jwt as _jwt
                    try:
                        payload = _jwt.decode(
                            token, session.hub_pk_pem, algorithms=["EdDSA"],
                            options={"verify_exp": False})
                        target = payload.get("target")
                        tid = payload.get("target_id", "")
                        if target == "user":
                            denylist.deny_user(tid)
                        elif target == "group":
                            # H4: previously dropped on the floor, so "suspend a
                            # group" was a hub-only gesture that no node enforced.
                            denylist.deny_group(tid)
                            self._drop_group_sessions(tid)
                        elif target == "jti":
                            denylist.deny_jti(tid)
                        else:
                            log.warning("Unknown revocation target: %r", target)
                    except Exception as e:
                        log.warning("Invalid revocation token: %s", e)

            ws_task = asyncio.create_task(hub.maintain_ws(
                on_incoming=on_incoming,
                on_revocation=on_revocation,
                on_webrtc_offer=on_webrtc_offer,
                group_ids=lambda: list((self._state.get("groups_ctx") or {}).keys()),
            ))
            self._tasks.append(ws_task)
            log.info("Hub WS task started")

            # 9. (removed in Phase 11.5) The per-group HTTP file API used to start here.
            # It served the Mesh Group Index and raw plaintext files on 0.0.0.0 with no
            # authentication, for private groups too — finding C1. Every client path now
            # goes through the MNP handshake (JWT + group claim + GEK proof).

            # 10. Update control API state (already running from step 2)
            self._state["groups_ctx"] = groups_ctx
            self._state["audit_store"] = self._audit_store
            self._state["bundle_store"] = self._bundle_store
            self._state["roster"] = self._roster
            self._state["node_user_id"] = session.user_id
            # The node's own Ed25519 key. Needed by `encrypt_chat_history`,
            # which seals migrated messages under a synthetic device of the
            # node's rather than pretending to hold a member's signing key.
            self._state["sk_node"] = keys.sk_ed25519
            self._state["webrtc"] = self._webrtc
            self._state["quic_server"] = self._quic_server
            self._state["hub"] = hub
            self._state["reload_fn"] = self._reload_config
            # Keyed by app, so `ops.set_app_directories` finds the right
            # sweep without knowing which apps exist — an app with nothing to
            # enrich simply has no entry.
            self._state["enrich_app_dirs_fns"] = {
                "video": self._enrich_video_root_now,
                "music": self._enrich_audio_root_now,
                "photo": self._enrich_photo_roots_now,
            }
            # Rotating a key has to reach every transport holding a copy of it,
            # and clearing the denylist has to reach the one the handshake
            # consults — so both are published rather than reachable only
            # through the object that happens to own them.
            self._state["denylist"] = self._denylist
            self._state["pk_x25519_raw"] = pk_x_raw
            self._state["sk_x25519_raw"] = sk_x_raw
            self._state["sk_ed25519"] = keys.sk_ed25519

            self._state["status"] = "running"
            log.info("Node ready — %d groups, WebRTC=%s, QUIC=%s",
                     len(groups_ctx),
                     "yes" if self._webrtc else "no",
                     "yes" if self._quic_server else "no")

            # 11. Background initial scan — files appear progressively.
            async def _bg_scan(indexer, name, gctx):
                await indexer.initial_scan()
                log.info("Background scan complete for %s: %d files",
                         name, indexer.index.count)
                # Swarm registration for public groups (after files are known).
                if gctx.get("visibility") == "public":
                    endpoint = f"webrtc:{self._config.node.quic_port}"
                    hashes = [e.id for e in gctx["index"].entries]
                    if hashes:
                        await self._register_swarm(hashes, endpoint)
                # initial_scan() itself never calls on_change (it predates
                # the concept — every existing caller only cared about the
                # scan finishing, not about notifying anyone) — but Videos
                # app enrichment (duration/thumb_hash/display_title/...)
                # hangs entirely off that callback (_broadcast_index_change).
                # Without this, every file already on disk at startup — the
                # common case, an existing library — would never get
                # enriched at all; only a file added later, while the node
                # is already running, would trigger it via the watchdog.
                await self._on_index_change(indexer)

            for idx, group_cfg in zip(self._indexers, self._config.groups):
                gctx = groups_ctx.get(group_cfg.id)
                if gctx:
                    self._tasks.append(asyncio.create_task(
                        _bg_scan(idx, group_cfg.name, gctx)))
                    self._tasks.append(asyncio.create_task(
                        self._progress_pusher(idx)))

            # 12. Wait for shutdown
            stop_event = asyncio.Event()
            loop = asyncio.get_event_loop()
            console_shutdown_done = None
            if sys.platform == "win32":
                # SIGBREAK: CTRL_BREAK_EVENT, how platform.autostart_end() asks
                # a per-user-mode daemon to stop gracefully instead of only
                # ever taskkill /F.
                for sig in (signal.SIGINT, signal.SIGTERM, signal.SIGBREAK):
                    signal.signal(sig, lambda *_: stop_event.set())
                # CTRL_CLOSE/LOGOFF/SHUTDOWN reach no Python signal at all --
                # see platform.install_console_close_handler for why this is
                # a separate mechanism rather than another signal.signal() line.
                from meshbay_node.platform import install_console_close_handler
                console_shutdown_done = install_console_close_handler(loop, stop_event)
            else:
                for sig in (signal.SIGINT, signal.SIGTERM):
                    loop.add_signal_handler(sig, stop_event.set)
            # Milestone 14.8: re-read node.toml without dropping connections.
            try:
                loop.add_signal_handler(
                    signal.SIGHUP,
                    lambda: spawn(self._reload_config()))
            except (NotImplementedError, AttributeError):
                pass          # no SIGHUP on Windows; `reload` says so there
            await stop_event.wait()

        await self._shutdown()
        if console_shutdown_done is not None:
            # Releases the console-control handler's blocking wait (see
            # platform.install_console_close_handler) so it can return and
            # let Windows actually end the process for CTRL_CLOSE/LOGOFF/
            # SHUTDOWN, now that cleanup is genuinely done rather than just
            # started.
            console_shutdown_done.set()

    async def _reload_config(self) -> None:
        """
        Re-read node.toml and reconcile groups.

        Handles root changes on existing groups, hot-loads new groups, and
        tears down removed groups. Existing connections are untouched: a member
        watching a film keeps watching it.

        Serialised by _reload_lock: fire-and-forget reloads from config-mutating
        endpoints can overlap with the wizard's explicit /api/reload call,
        and two concurrent hot-loads of the same group corrupt the runtime state.

        The reload belongs to the node, never to whoever asked for it. A root
        added from a browser reaches here through the operator's WebRTC session,
        whose tasks are all cancelled when that session closes — and on
        2026-09-14 one closed 47 s into the scan of a 900 GB root. The reload
        died without a line in the log, the new root was in node.toml and in
        the indexer but never in the group's context, and nothing ever tried
        again: the lock was free, and nobody was waiting on it. So the work runs
        in a task of its own, and a caller that goes away only stops waiting.
        """
        await asyncio.shield(spawn(self._reload_config_locked(), what="config reload"))

    async def _reload_config_locked(self) -> None:
        try:
            async with self._reload_lock:
                await self._reload_config_inner()
        except asyncio.CancelledError:
            log.warning("Config reload cancelled before it finished — the node may "
                        "be serving part of the previous configuration until the "
                        "next reload")
            raise

    async def _reload_config_inner(self) -> None:
        log.info("Reloading config from %s", self._config_path)
        try:
            fresh = load_config(self._config_path)
        except Exception as e:
            log.error("Reload failed, keeping the running config: %s", e)
            return

        groups_ctx = self._state.get("groups_ctx", {})
        hosted = set(groups_ctx)
        incoming = {g.id for g in fresh.groups if g.id}

        # ── Root changes on existing groups ──────────────────────────────
        changed = 0
        for group_cfg in fresh.groups:
            ctx = groups_ctx.get(group_cfg.id)
            if not ctx:
                continue
            try:
                roots = await self._build_roots(group_cfg)
            except RootError as e:
                log.error("Group %r: %s — keeping the roots already loaded",
                          group_cfg.name, e)
                continue
            # `writable` and `removable` are in the comparison because an
            # operator editing node.toml by hand and reloading is a supported
            # way to change them, and a set compared on name and path alone
            # reports "nothing changed" for exactly that edit.
            if _root_shape(ctx["roots"]) == _root_shape(roots):
                continue
            await off_disk(roots, roots.refresh_availability)
            indexer = next((i for i in self._indexers
                            if i.group_id == group_cfg.id), None)
            if indexer is None:
                continue
            log.info("Group %r roots changed: %s", group_cfg.name,
                     ", ".join(f"{r.name}={r.path}" for r in roots))
            # The new set is what the node serves from this moment, and the
            # scan of an added root is not waited for. Awaiting it here held
            # `_reload_lock` and the old set for as long as the scan ran —
            # hours for a large drive — so every file request under the new
            # root found no root to resolve against, and any op answering with
            # the live table (a writable/removable toggle) showed the directory
            # gone from the operator's settings.
            ctx["roots"] = roots
            await indexer.retarget(roots, wait=False)
            changed += 1

        # ── Hot-load new groups ──────────────────────────────────────────
        added_names = []
        sk_ed = self._state.get("sk_ed25519")
        sk_x_raw = self._state.get("sk_x25519_raw")
        pk_x_raw = self._state.get("pk_x25519_raw")
        node_user_id = self._state.get("node_user_id")
        data_dir = fresh.data_dir

        for group_cfg in fresh.groups:
            if group_cfg.id in hosted:
                continue
            if not group_cfg.id or not group_cfg.roots:
                log.warning("New group %r has no id or roots — skipping",
                            group_cfg.name)
                continue
            if not sk_ed:
                log.warning("Cannot hot-load %r — signing key not available",
                            group_cfg.name)
                continue

            try:
                roots = await self._build_roots(group_cfg)
            except RootError as e:
                log.error("New group %r: %s — skipping", group_cfg.name, e)
                continue
            await off_disk(roots, roots.refresh_availability)

            gek = None
            if sk_x_raw and pk_x_raw:
                gek = await self._load_gek(
                    group_cfg.id, node_user_id, sk_x_raw, pk_x_raw)
                if gek:
                    log.info("GEK loaded for new group %s", group_cfg.id[:8])

            scan_settings = (
                await self._roster.scan_settings(group_cfg.id)
                if self._roster else {
                    "reconcile_interval_secs": DirectoryIndexer.DEFAULT_RECONCILE_SECS,
                    "debounce_secs": DirectoryIndexer.DEFAULT_DEBOUNCE_SECS,
                })

            indexer = DirectoryIndexer(
                roots=roots,
                group_id=group_cfg.id,
                sk_node=sk_ed,
                gek=gek,
                on_change=self._on_index_change,
                on_root_ejected=self._eject_persister(group_cfg.id),
                cache=self._index_cache,
                reconcile_secs=scan_settings["reconcile_interval_secs"],
                debounce_secs=scan_settings["debounce_secs"],
            )
            # Registered *before* start() runs its (blocking, possibly very
            # long — see the StarWars benchmark) initial scan, specifically
            # so /api/groups/{id}/index-status can see indexer.progress
            # while a brand-new group is still scanning — this is the one
            # group state that must stay visible during the very window the
            # group is not yet authorized for member connections (below).
            self._indexers.append(indexer)
            self._state["indexes"][group_cfg.id] = indexer.index
            self._state["indexers"][group_cfg.id] = indexer

            await indexer.start()

            data_dir.mkdir(parents=True, exist_ok=True)
            chat_db = data_dir / group_cfg.id[:16] / "chat.db"
            store = ChatStore(db_path=chat_db)
            await store.open()
            self._chat_stores[group_cfg.id] = store

            new_ctx = {
                "gek": gek,
                "roots": roots,
                "index": indexer.index,
                "progress": indexer.progress,
                "note_activity": indexer.note_activity,
                "record_upload": indexer.record_upload,
                "reconcile_interval_secs": scan_settings["reconcile_interval_secs"],
                "debounce_secs": scan_settings["debounce_secs"],
                "visibility": group_cfg.visibility,
                "join_policy": group_cfg.join_policy,
                "enabled_apps": (
                    await self._roster.enabled_apps(group_cfg.id)
                    if self._roster else list(Roster.DEFAULT_APPS)),
                "transfer_limits": (
                    await self._roster.transfer_limits(group_cfg.id)
                    if self._roster else {}),
                **(await self._app_directories_ctx(group_cfg.id)),
                "chat_link_preview": (
                    await self._roster.chat_link_preview(group_cfg.id)
                    if self._roster else True),
                "search_listed": (
                    await self._roster.search_listed(group_cfg.id)
                    if self._roster else True),
                "chat_epoch": await self._ensure_chat_epoch(group_cfg.id),
                "tmdb_enabled": (
                    await self._roster.tmdb_enabled(group_cfg.id)
                    if self._roster else True),
                "musicbrainz_enabled": (
                    await self._roster.musicbrainz_enabled(group_cfg.id)
                    if self._roster else True),
                "chat_store": store,
            }
            groups_ctx[group_cfg.id] = new_ctx

            if self._webrtc:
                self._webrtc._ctx["groups"][group_cfg.id] = new_ctx
            log.info("Hot-loaded group %s (%s, %d roots)",
                     group_cfg.name, group_cfg.id[:8], len(roots))
            added_names.append(group_cfg.name)
            # The initial scan above already ran to completion (indexer.start()
            # is not deferred here), so this only matters for whatever scans
            # this group as time goes on — a root added later, reconcile
            # picking one back up.
            self._tasks.append(asyncio.create_task(self._progress_pusher(indexer)))

        # ── Tear down removed groups ─────────────────────────────────────
        removed_names = []
        for gid in hosted - incoming:
            indexer = next((i for i in self._indexers
                            if i.group_id == gid), None)
            if indexer:
                try:
                    await indexer.stop()
                except Exception:
                    pass
                self._indexers.remove(indexer)
            store = self._chat_stores.pop(gid, None)
            if store:
                try:
                    await store.close()
                except Exception:
                    pass
            # No per-group index cache to close here (2026-08-25): the
            # (path, size, mtime) -> hash cache is now one shared instance,
            # open for the life of the daemon, since another group may still
            # reference the same physical folder — see indexer/cache.py.
            pending = self._pending_broadcasts.pop(gid, None)
            if pending:
                pending.cancel()
            self._last_broadcast_snapshot.pop(gid, None)
            self._state["indexes"].pop(gid, None)
            self._state["indexers"].pop(gid, None)
            old_name = gid[:8]
            for g_cfg in self._config.groups:
                if g_cfg.id == gid:
                    old_name = g_cfg.name
                    break
            groups_ctx.pop(gid, None)
            if self._webrtc and self._webrtc._ctx.get("groups") is not groups_ctx:
                self._webrtc._ctx["groups"].pop(gid, None)
            log.info("Unloaded group %s (%s)", old_name, gid[:8])
            removed_names.append(old_name)

        self._config = fresh
        self._state["config"] = fresh
        self._state["groups"] = [g.name for g in fresh.groups]
        log.info("Reload complete — %d re-rooted, %d added, %d removed",
                 changed, len(added_names), len(removed_names))

        if (added_names or removed_names) and self._hub:
            gids = list((self._state.get("groups_ctx") or {}).keys())
            await self._hub.update_ws_groups(gids)

    async def _login_with_retry(self, hub: HubClient):
        """Login to hub, retrying if the node key hasn't been linked yet."""
        import httpx as _httpx
        while True:
            try:
                return await hub.startup(endpoint_hint=None)
            except _httpx.HTTPStatusError as e:
                body = e.response.text if hasattr(e.response, 'text') else ''
                # Any 401 here needs a human at a browser, and the operator needs
                # this daemon alive to read its public key (via `meshbay-node
                # status` or the desktop client, both of which query the control
                # API). Exiting would strand them — which is exactly what
                # happened when a node was started before its owner had
                # registered.
                if e.response.status_code == 401:
                    if "No node key" in body:
                        self._state["status"] = "waiting_for_node_key"
                        log.warning(
                            "Node key not linked. Get it from `meshbay-node "
                            "status` and paste it in Settings > Link Node on %s "
                            "(the desktop client links it automatically). "
                            "Retrying in 5s...",
                            self._config.hub.url,
                        )
                    else:
                        self._state["status"] = "waiting_for_account"
                        log.warning(
                            "Hub rejected the node credentials for user %r. "
                            "Register that account on %s first, then link this "
                            "node's key. Retrying in 5s...",
                            self._config.hub.username, self._config.hub.url,
                        )
                    await asyncio.sleep(5)
                else:
                    raise
            except Exception as e:
                log.warning("Hub login failed: %s — retrying in 10s", e)
                await asyncio.sleep(10)

    async def _load_gek(
        self,
        group_id: str,
        node_user_id: str,
        sk_x_raw: bytes,
        pk_x_raw: bytes,
    ) -> bytes | None:
        """Load GEK from local bundle store (node-only, hub never touches crypto)."""
        from meshbay_common.crypto import unwrap_gek_aes

        if not self._bundle_store:
            return None

        # Try node-specific bundle first (stored by init_gek for daemon reload),
        # then fall back to operator's user bundle (legacy / pre-dual-key)
        for user_key in [f"_node_{node_user_id}", node_user_id]:
            bundle = await self._bundle_store.fetch(group_id, user_key)
            if not bundle:
                continue
            try:
                gek = unwrap_gek_aes(bundle, sk_x_raw, pk_x_raw)
                log.info("GEK loaded from local bundle store for group %s (key=%s)",
                         group_id[:8], user_key[:16])
                return gek
            except Exception as e:
                log.debug("Failed to unwrap GEK bundle (key=%s): %s", user_key[:16], e)

        log.warning("No unwrappable GEK bundle found for group %s", group_id[:8])
        return None

    async def _ensure_chat_epoch(self, group_id: str) -> int:
        """
        The group's current chat epoch, opening the first one if it has none.

        Chat is always encrypted, so a group with no epoch key is a group in
        which nobody can say anything. Attaching one is the node's job and
        happens here rather than on the first message: a failure at start-up is
        in the log the operator is already reading, and a failure on someone's
        first message is a chat that mysteriously refuses them.

        Never fatal. A group whose epoch cannot be opened keeps every other
        function — files, video, the index — and only its chat is unusable,
        which is strictly better than refusing to host the group at all.
        """
        if not self._bundle_store:
            return 0
        try:
            epoch = await self._bundle_store.latest_chat_epoch(group_id)
            if epoch:
                return epoch
            from meshbay_node import ops

            return (await ops.open_chat_epoch(self._state, group_id))["epoch"]
        except Exception as e:
            log.error("chat: no epoch key for group %s (%s) — chat is "
                      "unusable in this group until this is fixed",
                      group_id[:8], e)
            return 0

    async def _reap_partial_uploads(self, interval: float = 3600.0,
                                    first_delay: float = 60.0) -> None:
        """
        Delete `.part` files that no upload will ever finish.

        An upload interrupted for good leaves its partial file behind, and
        nothing else ever looks at it: `.part` is not an index entry, so it is
        invisible to every group member and to the operator's own file list. One
        abandoned film is a gigabyte of their disk, kept for ever.

        Two conditions, both required, and `uploads.orphaned_parts` is where
        they are stated and tested. What this adds is the walk and the deletion,
        and one rule of its own: it runs a minute after start rather than at
        once, so a client reconnecting to finish an upload that outlived a node
        restart is not raced by the janitor that would have deleted it — the age
        threshold makes that impossible in practice, and doing it anyway costs a
        minute.

        `interval` and `first_delay` are parameters so a test can drive this
        without waiting an hour.
        """
        await asyncio.sleep(first_delay)
        while True:
            try:
                self._reap_once()
            except Exception as exc:   # never let the janitor kill the node
                log.warning("Reaping partial uploads failed: %s", exc)
            await asyncio.sleep(interval)

    async def _purge_audit_log(self, interval: float = 86400.0) -> None:
        """
        Delete audit entries past the retention period, at start then daily.

        `AuditStore.cleanup` leaves scheduling to its caller; until this loop
        nobody called it and a node kept every entry for ever. `interval` is a
        parameter so a test can drive this without waiting a day.
        """
        while True:
            try:
                if self._audit_store:
                    deleted = await self._audit_store.cleanup()
                    if deleted:
                        log.info("Purged %d audit entries older than %d days",
                                 deleted, AUDIT_RETENTION_DAYS)
            except Exception as exc:   # never let the janitor kill the node
                log.warning("Purging the audit log failed: %s", exc)
            await asyncio.sleep(interval)

    def _reap_once(self, now: float | None = None) -> int:
        """One pass over every group. Returns how many files were deleted."""
        groups = (self._webrtc._ctx.get("groups") or {}) if self._webrtc else {}
        when = time.time() if now is None else now
        deleted = 0
        for gid, ctx in groups.items():
            roots = ctx.get("roots")
            if roots is None:
                continue
            store = ctx.get("partial_uploads")
            live = store.live_paths() if store is not None else set()
            for path in uploads_mod.orphaned_parts(
                    uploads_mod.find_parts(roots.roots), live, when):
                try:
                    size = path.stat().st_size
                    path.unlink()
                except OSError as exc:
                    log.warning("Could not remove abandoned upload %s: %s",
                                path.name, exc)
                    continue
                deleted += 1
                log.info("Removed abandoned upload %s (%d bytes, group %s)",
                         path.name, size, gid[:8])
        return deleted

    async def _progress_pusher(self, indexer: DirectoryIndexer,
                               interval: float = 2.0) -> None:
        """
        Watches indexer.progress and pushes a light INDEX_PROGRESS message to
        this group's connected peers — never the index itself, that stays
        _on_index_change's job. Runs for the node's whole lifetime: a scan
        can start from several places (initial scan, a root added later,
        reconcile picking a root back up), and this only needs to notice the
        flag, not why it changed.

        The final push at the False transition is what lets a presence dot
        reliably turn back off on an already-connected client — the
        handshake ack only covers the moment of connecting. `interval` is a
        parameter (not a bare constant) only so a test can drive this loop
        without waiting on the real 2s cadence.
        """
        was_busy = False
        while True:
            await asyncio.sleep(interval)
            progress = indexer.progress
            # A root waiting for the scan lock is work the operator is owed a
            # sight of, even in the gap where nothing is walking yet.
            busy = progress.scanning or bool(progress.queued)
            if busy or was_busy:
                self._push_index_progress(indexer.group_id, progress)
            was_busy = busy

    def _push_index_progress(self, group_id: str, progress) -> None:
        if not self._webrtc:
            return
        # Deliberately NOT sealed, unlike index_sync/index_delta (decision D3).
        # Counters only — never a path, never a filename, see IndexProgress in
        # indexer.py — pushed every couple of seconds for the whole length of a
        # scan. Sealing it would buy an attacker's rough estimate of a library's
        # size and cost a key derivation and a decrypt per push. If a field that
        # names anything is ever added here, that trade is void and this message
        # joins the other two. `kind` is one of four fixed words; the root under
        # way is a position in the roots table the member already opened from
        # the sealed index, and the roots waiting are a count, never names.
        msg = {
            "type": MNP.INDEX_PROGRESS,
            "v": MNP_VERSION,
            "group_id": group_id,
            "scanning": progress.scanning,
            "scanned_bytes": progress.scanned_bytes,
            "total_bytes": progress.total_bytes,
            "files_done": progress.files_done,
            "files_total": progress.files_total,
            "kind": progress.kind,
            "root_pos": progress.root_pos,
            "queued": len(progress.queued),
        }
        pushed = 0
        for session in list(self._webrtc._sessions.values()):
            if session._group_id == group_id:
                try:
                    session._send(msg)
                    pushed += 1
                except Exception:
                    pass
        if pushed:
            log.debug("Index progress pushed to %d peer(s) for group %s",
                      pushed, group_id[:8])

    async def _build_roots(self, group_cfg) -> RootSet:
        """
        Build a group's RootSet from node.toml, with the ejected state restored.

        node.toml carries configuration (`writable`, `removable`); the roster
        carries the runtime answer to "is this drive ejected right now". They
        are merged here, in the one place every caller goes through, because a
        root that quietly comes back available across a restart is exactly the
        surprise unplug that eject exists to survive.
        """
        specs = [asdict(r) for r in group_cfg.roots]
        if self._roster:
            ejected = await self._roster.ejected_roots(group_cfg.id)
            if ejected:
                for spec in specs:
                    name = spec.get("name") or Path(spec.get("path", "")).name
                    if fold(name) in ejected:
                        spec["ejected"] = True
        return RootSet.build(specs)

    # Every application that keeps directories. This is the one list, and it
    # lives here because the daemon is what wires a group's context: `roster.py`,
    # `ops.py` and the rest must name no application at all — that is the
    # property the reference app exists to demonstrate
    # (`test_helloworld_proves_the_plugin_claim.py`).
    #
    # Not derived from `enabled_apps`: the context is read once at load, and an
    # application enabled later must not find its own setting missing.
    #
    # The handshake ack does **not** get a copy of this. It emits whatever
    # `<app>_directories` the context holds, so the two cannot drift — a copy
    # lived in `webrtc_server.py` until 2026-09-10 and had already lost
    # `helloworld`, which made the app that proves a new one needs no
    # special-casing the single app whose directories never reached a client.
    APP_DIR_KEYS = ("video", "music", "photo", "chat", "helloworld")

    async def _app_directories_ctx(self, group_id: str) -> dict:
        """
        Each app's configured directories, plus the second name an app is
        also published under where something reads one (`chat_directory`).

        Derived here rather than stored, so the two can never disagree.
        """
        dirs = {}
        for app in self.APP_DIR_KEYS:
            dirs[f"{app}_directories"] = (
                await self._roster.app_directories(group_id, app)
                if self._roster else [])
        aliases = {}
        for app in self.APP_DIR_KEYS:
            alias = Roster.ctx_alias(app, dirs[f"{app}_directories"])
            if alias:
                aliases[alias[0]] = alias[1]
        return {**dirs, **aliases}

    def _eject_persister(self, group_id: str):
        """`on_root_ejected` bound to one group, for that group's indexer."""
        async def persist(root_name: str, ejected: bool) -> None:
            if self._roster:
                await self._roster.set_root_ejected(
                    group_id, root_name, ejected,
                    set_by=self._state.get("node_user_id", ""))
        return persist

    async def _on_index_change(self, indexer: DirectoryIndexer) -> None:
        """
        Called when a DirectoryIndexer detects file changes — once per
        debounced watchdog event, so dropping N files into a watched folder
        calls this N times in quick succession. Coalesces those into one
        broadcast (_broadcast_index_change) rather than one push per file:
        the timer is reset on every call and only fires once calls stop
        arriving for _broadcast_coalesce_secs.
        """
        group_id = indexer.group_id
        loop = asyncio.get_event_loop()
        pending = self._pending_broadcasts.pop(group_id, None)
        if pending:
            pending.cancel()

        def fire() -> None:
            self._pending_broadcasts.pop(group_id, None)
            spawn(self._broadcast_index_change(indexer))

        self._pending_broadcasts[group_id] = loop.call_later(
            self._broadcast_coalesce_secs, fire)

    async def _broadcast_index_change(self, indexer: DirectoryIndexer) -> None:
        """
        The actual push, run once per coalesced burst. Sends a full
        INDEX_SYNC the first time a group is ever broadcast (no previous
        snapshot to diff against — the client's own first fetchIndex() call
        already covers that case) and an INDEX_DELTA every time after,
        computed against the last thing this method actually sent.
        """
        group_id = indexer.group_id
        idx = indexer.index
        log.info("Index changed for group %s: %d files (v%d)",
                 group_id[:8], idx.count, idx.version)

        prev = self._last_broadcast_snapshot.get(group_id)
        delta = None
        previous = None
        if prev is not None:
            prev_version, prev_entries = prev
            previous = GroupIndex._snapshot(
                idx.group_id, idx.sk_node, idx.gek, prev_version, prev_entries)
            delta = idx.diff(previous)
        self._last_broadcast_snapshot[group_id] = (idx.version, idx.entries_by_id())

        # Videos app (docs/MESHBAY_DESIGN.md §6.5): schedule async technical
        # probe + title parse + thumbnail generation for every newly-seen
        # video entry under the group's configured video_root. Never blocks
        # this broadcast — enrichment fields arrive later as their own
        # INDEX_DELTA update (_on_enriched below).
        new_entries = delta.additions if delta is not None else list(idx.entries)

        # A root that was ejected and plugged back in, or that fell off and
        # re-mounted, has had its entries thrown away and rebuilt from disk
        # (`indexer._drop_root_entries`). The rebuilt entry has the same
        # content-hash id and none of the enrichment fields, so the diff above
        # reports neither an addition nor a deletion — and `_enriched_attempted`
        # still says "done" for a file whose album and cover no longer exist.
        # Found live: a Music library came back with its files and without its
        # albums, and stayed that way, because only a restart (which starts
        # with no snapshot, making every entry an addition) could clear either
        # gate. Treated here as what it is — those entries are new again.
        rebuilt_ids = indexer.drain_rescanned_ids()
        if rebuilt_ids:
            rebuilt = [e for e in idx.entries if e.id in rebuilt_ids]
            for entry in rebuilt:
                self._enriched_attempted.discard((group_id, entry.id))
            seen = {e.id for e in new_entries}
            new_entries = new_entries + [e for e in rebuilt if e.id not in seen]
        spawn(self._enrich_new_video_entries(indexer, new_entries))
        # Music app (docs/MESHBAY_DESIGN.md §9.8): same shape, gated on
        # audio_root exactly like video_root above (added later — the original
        # "no root, whole shared tree" call didn't hold up).
        spawn(self._enrich_new_audio_entries(indexer, new_entries))
        # Photos app (docs/MESHBAY_DESIGN.md §9.9): same shape, gated on
        # photo_roots (a list, not a single string).
        spawn(self._enrich_new_photo_entries(indexer, new_entries))

        # A rename/move changes the very filename (or season folder) that
        # docs/MESHBAY_DESIGN.md §9.7's title-parse read
        # display_title/season/episode from, but leaves the file's content —
        # and so its id and everything ffprobe/thumbnailing already found —
        # untouched. Only entries
        # whose name or path actually differ from the last broadcast get a
        # fresh pass; an update that is enrichment's own field-fill
        # (duration/thumb_hash/... landing via _on_enriched below) leaves
        # name/path alone and must not re-trigger itself forever.
        if delta is not None and delta.updates and previous is not None:
            spawn(
                self._reenrich_renamed_video_entries(indexer, delta.updates, previous))
            spawn(
                self._reenrich_renamed_audio_entries(indexer, delta.updates, previous))
            spawn(
                self._reenrich_renamed_photo_entries(indexer, delta.updates, previous))

        # Videos/Music/Photos apps: a file that leaves the index also loses
        # its thumbnail/cover and file->tmdb/file->mbid mapping — the "real
        # deletion obligation" docs/MESHBAY_DESIGN.md §6.5 calls out
        # explicitly rather than leaving implicit (docs/MESHBAY_DESIGN.md §9.8
        # follows the same rule). tmdb_meta/mbid_meta rows are left alone
        # (shared across files).
        #
        # Found live (docs/MESHBAY_DESIGN.md §9.9): a root removed and a new
        # one added for the identical content (an operator renaming/relocating a
        # shared folder) pruned the thumbnail here — correctly, the content
        # is gone from *this* root — but left the hash in
        # `_enriched_attempted`, which is never otherwise cleared. The same
        # bytes reappearing under the new root's path were then permanently
        # skipped: "already attempted" was true forever, for a thumbnail
        # that no longer existed. Discarding the attempt alongside the
        # cache entry is what makes pruning actually reversible — the next
        # sweep re-enriches it exactly as if it were new, which content
        # that is content-addressed and simply moved effectively is.
        if delta is not None and delta.deletions and self._media_cache:
            for file_id in delta.deletions:
                self._enriched_attempted.discard((indexer.group_id, file_id))
                # media_cache.db is node-wide, keyed by content hash — a file
                # shared into two groups is one row there, same reasoning as
                # `_enriched_attempted`'s own docstring above. This group's
                # copy is genuinely gone (that's what a deletion delta is),
                # but another group may still hold the same content: only
                # prune once *no* group's index has this file_id any more,
                # or the surviving group pays for a redundant re-fetch/
                # re-probe/re-thumbnail for content it never actually lost.
                still_referenced = any(
                    i.index.get_entry(file_id) is not None for i in self._indexers)
                if not still_referenced:
                    spawn(self._media_cache.prune_file(file_id))

        # 11.5 — Push to connected WebRTC peers in this group
        if self._webrtc:
            peers = [s for s in list(self._webrtc._sessions.values())
                     if s._group_id == group_id]
            # Both messages are sealed under a GEK-derived subkey, so building one
            # needs a key. A group without one has no peers to push to either — the
            # node refuses every handshake while the GEK is None (NS8) — so this is
            # "nobody is listening", not a case to send in clear for.
            if peers and idx.gek:
                msg = (index_delta_message(idx, delta, indexer.roots)
                       if delta is not None
                       else index_sync_message(idx, indexer.roots))
                pushed = 0
                for session in peers:
                    try:
                        session._send(msg)
                        pushed += 1
                    except Exception:
                        pass
                if pushed:
                    log.info("Index %s pushed to %d WebRTC peers",
                             "delta" if delta is not None else "sync", pushed)

        # 11.9 — Register file hashes with hub swarm table (public groups only, H7)
        group_cfg = next(
            (g for g in self._config.groups if g.id == group_id), None)
        if (self._hub and self._state.get("endpoint_hint")
                and group_cfg and group_cfg.visibility == "public"):
            # Only the newly added hashes once there is a delta to know them
            # from — registering the whole library again on every change is
            # the same O(changes x library size) cost the delta above exists
            # to avoid.
            hashes = ([e.id for e in delta.additions] if delta is not None
                     else [e.id for e in idx.entries])
            if hashes:
                endpoint = f"webrtc:{self._config.node.quic_port}"
                spawn(self._register_swarm(hashes, endpoint))

    def _drop_group_sessions(self, group_id: str) -> None:
        """Close live sessions for a revoked group (H4)."""
        if not self._webrtc or not group_id:
            return
        for session in list(self._webrtc._sessions.values()):
            if session._group_id == group_id:
                spawn(session.close())
                log.info("Dropped session for revoked group %s", group_id[:8])

    async def _register_swarm(self, hashes: list[str], endpoint: str) -> None:
        try:
            n = await self._hub.register_swarm(hashes, endpoint)
            log.info("Swarm: registered %d/%d hashes", n, len(hashes))
        except Exception as e:
            log.warning("Swarm registration failed: %s", e)

    async def _shutdown(self) -> None:
        log.info("Shutting down...")
        self._state["status"] = "stopping"

        for handle in self._pending_broadcasts.values():
            handle.cancel()
        self._pending_broadcasts.clear()

        for task in self._tasks:
            task.cancel()
        for task in self._tasks:
            try:
                await task
            except (asyncio.CancelledError, Exception):
                pass

        if self._webrtc:
            await self._webrtc.close_all()

        if self._audit_store:
            await self._audit_store.close()

        if self._bundle_store:
            await self._bundle_store.close()

        if self._tmdb_client:
            await self._tmdb_client.close()

        if self._musicbrainz_client:
            await self._musicbrainz_client.close()

        if self._media_cache:
            await self._media_cache.close()

        if self._roster:
            await self._roster.close()

        for store in self._chat_stores.values():
            await store.close()

        if self._index_cache:
            await self._index_cache.close()

        for indexer in self._indexers:
            await indexer.stop()

        # The thread each group's roots are read from (roots.py `io_executor`).
        # Created on the first read, so a group nobody downloaded from never
        # started one and this is a no-op for it.
        groups = self._webrtc._ctx.get("groups") if self._webrtc else None
        for group in (groups or {}).values():
            roots = group.get("roots")
            if roots is not None:
                roots.close_io()

        if self._quic_server:
            await self._quic_server.stop()

        token_file = getattr(self, "_ui_token_file", None)
        if token_file is not None:
            token_file.unlink(missing_ok=True)

        log.info("Node stopped")


# ── Entry point ───────────────────────────────────────────────────────────────

def main() -> None:
    args = start()
    if run(args):
        return

    cfg = load_config(args.config or DEFAULT_CONFIG_PATH)
    if not cfg.hub.username:
        print("Error: hub.username not set in config. Run: meshbay-node init")
        sys.exit(1)

    from meshbay_node.platform import check_media_tools
    try:
        check_media_tools(cfg.node.ffmpeg_path, cfg.node.ffprobe_path)
    except RuntimeError as e:
        print(f"Error: {e}")
        sys.exit(1)

    daemon = NodeDaemon(cfg, Path(args.config or DEFAULT_CONFIG_PATH))
    asyncio.run(daemon.run())


if __name__ == "__main__":
    main()