aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/indexer/group_index.py
blob: 85bc13e890660d9147683905ca51f0c411664711 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
"""
Mesh Group Index — the file listing for one group, and the deltas between versions.

What travels on the wire is built from it by `transport/wire.py` and sealed under
the group key (`meshbay_common.groupbox`); this module holds no encoding of its own.

Delta format:
  {base_version, version, additions: [...], deletions: [id, ...]}
"""

import logging
from dataclasses import dataclass, field

from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
from meshbay_common.protocol import IndexDelta, IndexEntry

log = logging.getLogger(__name__)


@dataclass
class GroupIndex:
    """
    Mesh Group Index for one group.

    Usage:
        idx = GroupIndex(group_id="...", sk_node=sk, gek=gek_bytes)
        idx.add_entry(entry)
    """

    group_id: str
    sk_node:  Ed25519PrivateKey
    gek:      bytes | None = None   # None → public group (no encryption)
    version:  int = 1
    # The group's roots and whether each is readable right now, as the indexer
    # last described them. A member needs this to tell "temporarily unavailable"
    # from "deleted" — a distinction the entries alone cannot carry, since an
    # unavailable root's files are still listed. What members receive is built
    # from the RootSet by `transport/wire.py`, not from this copy.
    roots:    list = field(default_factory=list)
    _entries: dict = field(default_factory=dict, repr=False)  # id → IndexEntry

    # ── Entry management ──────────────────────────────────────────────────────

    def add_entry(self, entry: IndexEntry) -> None:
        self._entries[entry.id] = entry

    def remove_entry(self, file_id: str) -> bool:
        return self._entries.pop(file_id, None) is not None

    def get_entry(self, file_id: str) -> IndexEntry | None:
        return self._entries.get(file_id)

    @property
    def entries(self) -> list[IndexEntry]:
        return list(self._entries.values())

    @property
    def count(self) -> int:
        return len(self._entries)

    def entries_by_id(self) -> dict:
        """A snapshot copy, for diff() to compare a later version against —
        see daemon.py _on_index_change, the only caller."""
        return dict(self._entries)

    @classmethod
    def _snapshot(cls, group_id: str, sk_node: Ed25519PrivateKey, gek: bytes | None,
                  version: int, entries_by_id: dict) -> "GroupIndex":
        """
        A lightweight stand-in for diff()'s `previous` argument — never
        serialized or sent anywhere, just a comparison point built from an
        earlier entries_by_id() snapshot rather than a live GroupIndex.
        """
        idx = cls(group_id=group_id, sk_node=sk_node, gek=gek, version=version)
        idx._entries = dict(entries_by_id)
        return idx

    # ── Delta ─────────────────────────────────────────────────────────────────

    def diff(self, previous: "GroupIndex") -> IndexDelta:
        """
        Compute what changed since a previous version of this index.

        A shared id whose entry object now compares unequal (field-by-field,
        via IndexEntry's dataclass-generated __eq__) is an update, not an
        addition — the Videos app's async enrichment (duration, thumb_hash,
        title, ...) replaces an existing entry's fields after the fact via
        `add_entry`, which never introduces a new id. This only works
        because that replacement always constructs a *new* IndexEntry object
        (`dataclasses.replace`, never in-place attribute mutation) — mutating
        the same object in place would also mutate `previous`'s copy, since
        entries_by_id() is a shallow dict copy, and the two would always
        compare equal.
        """
        prev_ids = set(previous._entries)
        curr_ids = set(self._entries)

        additions = [self._entries[i] for i in curr_ids - prev_ids]
        deletions = list(prev_ids - curr_ids)
        updates = [
            self._entries[i] for i in curr_ids & prev_ids
            if self._entries[i] != previous._entries[i]
        ]

        return IndexDelta(
            base_version=previous.version,
            version=self.version,
            additions=additions,
            deletions=deletions,
            updates=updates,
        )