summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/indexer
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-08-26 00:40:25 +0200
committerChristophe Besson <cbesson@gmail.com>2026-08-26 00:40:25 +0200
commit704cfe37506fc5316c997b025004fbc9c4d2a47b (patch)
treebb51f6dc2ae395f56afc05e6a048a23d5b409fdf /packages/meshbay-node/src/meshbay_node/indexer
parent2af320ba4da49547176ef7e4c081956c33841958 (diff)
parent37d8d9c15c982f2da17b2fad4ea1a90613b560a6 (diff)
downloadmeshbay-704cfe37506fc5316c997b025004fbc9c4d2a47b.tar.gz
Merge branch 'feat/cross-group-file-cache'
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013XSohfUQQiaE77qyFLgSv3
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/indexer')
-rw-r--r--packages/meshbay-node/src/meshbay_node/indexer/cache.py42
1 files changed, 40 insertions, 2 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/cache.py b/packages/meshbay-node/src/meshbay_node/indexer/cache.py
index c26ddf1..31b9b15 100644
--- a/packages/meshbay-node/src/meshbay_node/indexer/cache.py
+++ b/packages/meshbay-node/src/meshbay_node/indexer/cache.py
@@ -1,5 +1,5 @@
"""
-MeshBay Node — persistent (path, size, mtime) -> hash cache, one per group.
+MeshBay Node — persistent (path, size, mtime) -> hash cache, one per node.
Without this, every node restart re-reads and re-hashes every file in every
root, even when nothing changed — measured at 23 minutes for a 114 GB library
@@ -10,6 +10,16 @@ It is a path-keyed accelerator only. The GroupIndex itself stays keyed by
content hash (see indexer.py's note on why two identical files are one
entry) — this cache never changes that, it only avoids recomputing a hash
that has not changed.
+
+**Shared by every group's DirectoryIndexer, one instance, one open
+connection** (2026-08-25) — an operator very often shares the same physical
+folder (a music library, a Séries drive) into more than one group, and a
+cache keyed purely by absolute path has no reason to care which group asked.
+It used to be opened once per group (`data_dir/{group_id}/index_cache.db`),
+which meant the second group to reference an already-fully-hashed multi-
+terabyte folder paid the same full read the first one did — exactly the cost
+this cache exists to avoid. The schema carries no group_id and never has;
+only daemon.py's wiring changed.
"""
import logging
@@ -40,7 +50,7 @@ class CachedEntry:
class IndexCache:
- """Async SQLite (path, size, mtime) -> hash cache for one group."""
+ """Async SQLite (path, size, mtime) -> hash cache, shared node-wide."""
def __init__(self, db_path: Path):
self._db_path = db_path
@@ -93,3 +103,31 @@ class IndexCache:
"type = excluded.type, added_at = excluded.added_at",
(path, mtime, size, hash, type, added_at))
await self._db.commit()
+
+ # ── Maintenance (node admin UI "prune index cache") ──────────────────────
+
+ async def count(self) -> int:
+ """Cheap — used for the dashboard stat, never for the prune decision
+ itself (that needs the actual paths, see all_paths)."""
+ async with self._db.execute("SELECT COUNT(*) FROM files") as cur:
+ row = await cur.fetchone()
+ return row[0] if row else 0
+
+ async def all_paths(self) -> list[str]:
+ """Every cached path, for a caller that decides staleness itself —
+ this cache has no notion of which paths are still claimed by a
+ group's roots, on purpose (see ops.prune_index_cache)."""
+ async with self._db.execute("SELECT path FROM files") as cur:
+ rows = await cur.fetchall()
+ return [row[0] for row in rows]
+
+ async def remove_many(self, paths: list[str]) -> None:
+ """Drop rows outright — used only for paths a caller has already
+ decided are gone for good. Losing one costs a rehash next time that
+ path is scanned, never a wrong answer (lookup() always re-validates
+ against a live stat())."""
+ if not paths:
+ return
+ await self._db.executemany(
+ "DELETE FROM files WHERE path = ?", [(p,) for p in paths])
+ await self._db.commit()