aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/indexer/cache.py
blob: 31b9b15cc5dbd9827827a396f2cabec31f7d5087 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
"""
MeshBay Node — persistent (path, size, mtime) -> hash cache, one per node.

Without this, every node restart re-reads and re-hashes every file in every
root, even when nothing changed — measured at 23 minutes for a 114 GB library
on a USB hard drive. This cache lets a scan skip the read entirely for a file
whose size and mtime still match what was hashed last time.

It is a path-keyed accelerator only. The GroupIndex itself stays keyed by
content hash (see indexer.py's note on why two identical files are one
entry) — this cache never changes that, it only avoids recomputing a hash
that has not changed.

**Shared by every group's DirectoryIndexer, one instance, one open
connection** (2026-08-25) — an operator very often shares the same physical
folder (a music library, a Séries drive) into more than one group, and a
cache keyed purely by absolute path has no reason to care which group asked.
It used to be opened once per group (`data_dir/{group_id}/index_cache.db`),
which meant the second group to reference an already-fully-hashed multi-
terabyte folder paid the same full read the first one did — exactly the cost
this cache exists to avoid. The schema carries no group_id and never has;
only daemon.py's wiring changed.
"""

import logging
from dataclasses import dataclass
from pathlib import Path

import aiosqlite

log = logging.getLogger(__name__)

_SCHEMA = """
CREATE TABLE IF NOT EXISTS files (
    path      TEXT PRIMARY KEY,
    mtime     REAL NOT NULL,
    size      INTEGER NOT NULL,
    hash      TEXT NOT NULL,
    type      TEXT NOT NULL,
    added_at  INTEGER NOT NULL
);
"""


@dataclass
class CachedEntry:
    hash: str
    type: str
    added_at: int


class IndexCache:
    """Async SQLite (path, size, mtime) -> hash cache, shared node-wide."""

    def __init__(self, db_path: Path):
        self._db_path = db_path
        self._db: aiosqlite.Connection | None = None

    async def open(self) -> None:
        self._db_path.parent.mkdir(parents=True, exist_ok=True)
        self._db = await aiosqlite.connect(str(self._db_path))
        await self._db.executescript(_SCHEMA)
        await self._db.commit()

    async def close(self) -> None:
        if self._db:
            await self._db.close()
            self._db = None

    async def __aenter__(self):
        await self.open()
        return self

    async def __aexit__(self, *_):
        await self.close()

    async def lookup(self, path: str, size: int, mtime: float) -> CachedEntry | None:
        """
        A cache hit requires an EXACT match on both size and mtime. A mtime
        touched without a content change is a false negative (an unnecessary
        rehash) — accepted, since the alternative (trusting a stale hash) is
        a silent wrong answer instead of an occasional wasted read.
        """
        async with self._db.execute(
                "SELECT hash, type, added_at FROM files "
                "WHERE path = ? AND size = ? AND mtime = ?",
                (path, size, mtime)) as cur:
            row = await cur.fetchone()
        return CachedEntry(hash=row[0], type=row[1], added_at=row[2]) if row else None

    async def put(self, path: str, size: int, mtime: float, hash: str,
                  type: str, added_at: int) -> None:
        """
        Written only once a file has been hashed in full — never partway
        through — so a crash mid-hash leaves no stale/partial row behind: the
        next scan simply finds no cache entry and hashes the file again.
        """
        await self._db.execute(
            "INSERT INTO files (path, mtime, size, hash, type, added_at) "
            "VALUES (?, ?, ?, ?, ?, ?) "
            "ON CONFLICT(path) DO UPDATE SET "
            "mtime = excluded.mtime, size = excluded.size, hash = excluded.hash, "
            "type = excluded.type, added_at = excluded.added_at",
            (path, mtime, size, hash, type, added_at))
        await self._db.commit()

    # ── Maintenance (node admin UI "prune index cache") ──────────────────────

    async def count(self) -> int:
        """Cheap — used for the dashboard stat, never for the prune decision
        itself (that needs the actual paths, see all_paths)."""
        async with self._db.execute("SELECT COUNT(*) FROM files") as cur:
            row = await cur.fetchone()
        return row[0] if row else 0

    async def all_paths(self) -> list[str]:
        """Every cached path, for a caller that decides staleness itself —
        this cache has no notion of which paths are still claimed by a
        group's roots, on purpose (see ops.prune_index_cache)."""
        async with self._db.execute("SELECT path FROM files") as cur:
            rows = await cur.fetchall()
        return [row[0] for row in rows]

    async def remove_many(self, paths: list[str]) -> None:
        """Drop rows outright — used only for paths a caller has already
        decided are gone for good. Losing one costs a rehash next time that
        path is scanned, never a wrong answer (lookup() always re-validates
        against a live stat())."""
        if not paths:
            return
        await self._db.executemany(
            "DELETE FROM files WHERE path = ?", [(p,) for p in paths])
        await self._db.commit()