summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/indexer
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-10-08 01:27:58 +0200
committerChristophe Besson <cbesson@gmail.com>2026-10-08 01:27:58 +0200
commitc27e04c88557716bbe8e42b9174ba9b07f90facf (patch)
treea1a56edbede2bba9bb0cc99b81b98570b8313915 /packages/meshbay-node/src/meshbay_node/indexer
parentcdd5fd52e981c4c59643e7dee705b6c55acae68e (diff)
downloadmeshbay-0.19.tar.gz
perf(node): sample 9 MB with the size above 9 MB, keep known ids0.19
hash_version 3: size + first 4 MB + last 4 MB + 1 MB at the middle, 5.5x faster cold on a USB disk than the 45 MB sample. The cache now serves a hit under whatever version it holds, so no existing id moves. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/indexer')
-rw-r--r--packages/meshbay-node/src/meshbay_node/indexer/cache.py13
-rw-r--r--packages/meshbay-node/src/meshbay_node/indexer/indexer.py33
2 files changed, 26 insertions, 20 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/cache.py b/packages/meshbay-node/src/meshbay_node/indexer/cache.py
index 5e328ef..afa4f85 100644
--- a/packages/meshbay-node/src/meshbay_node/indexer/cache.py
+++ b/packages/meshbay-node/src/meshbay_node/indexer/cache.py
@@ -112,17 +112,16 @@ class IndexCache:
async def __aexit__(self, *_):
await self.close()
- async def lookup(self, path: str, size: int, mtime: float,
- hash_version: int = 1) -> CachedEntry | None:
+ async def lookup(self, path: str, size: int, mtime: float) -> CachedEntry | None:
"""
- A cache hit requires an EXACT match on size, mtime AND hash_version.
- A v1 cached hash won't serve a v2 lookup for the same path — the file
- is re-hashed with the new algorithm instead.
+ A cache hit requires an EXACT match on size and mtime, and serves the
+ hash under whatever hash_version it was computed: a scheme change must
+ not change the id of a file the node already knows.
"""
async with self._db.execute(
"SELECT hash, type, added_at, hash_version FROM files "
- "WHERE path = ? AND size = ? AND mtime = ? AND hash_version = ?",
- (path, size, mtime, hash_version)) as cur:
+ "WHERE path = ? AND size = ? AND mtime = ?",
+ (path, size, mtime)) as cur:
row = await cur.fetchone()
return CachedEntry(hash=row[0], type=row[1], added_at=row[2],
hash_version=row[3]) if row else None
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/indexer.py b/packages/meshbay-node/src/meshbay_node/indexer/indexer.py
index f06d30f..4eeeec0 100644
--- a/packages/meshbay-node/src/meshbay_node/indexer/indexer.py
+++ b/packages/meshbay-node/src/meshbay_node/indexer/indexer.py
@@ -97,10 +97,14 @@ def _is_indexable_size(path: Path, size: int) -> bool:
_HASH_CHUNK = 8 * 1024 * 1024 # 8 MB streaming hash chunks
-_PARTIAL_THRESHOLD = 40 * 1024 * 1024 # files above this use partial-read hashing
-_PARTIAL_HEAD = 20 * 1024 * 1024
-_PARTIAL_TAIL = 20 * 1024 * 1024
-_PARTIAL_MID = 5 * 1024 * 1024
+# On a spinning disk the cost of a sample is its three seeks, not its bytes:
+# measured cold on USB, 45 MB took 516 ms a file, 9 MB 94 ms, 3 MB 79 ms. The
+# threshold is the sample size, so no file costs more to read than a sample.
+_PARTIAL_THRESHOLD = 9 * 1024 * 1024 # files above this use partial-read hashing
+_PARTIAL_HEAD = 4 * 1024 * 1024
+_PARTIAL_TAIL = 4 * 1024 * 1024
+_PARTIAL_MID = 1 * 1024 * 1024
+_PARTIAL_VERSION = 3
def _feed(hasher, f, nbytes: int) -> None:
@@ -115,6 +119,7 @@ def _feed(hasher, f, nbytes: int) -> None:
def _partial_hash(file_path: Path, size: int) -> str:
hasher = blake3.blake3()
+ hasher.update(size.to_bytes(8, "little"))
with open(long_path(file_path), "rb") as f:
_feed(hasher, f, _PARTIAL_HEAD)
f.seek(size - _PARTIAL_TAIL)
@@ -204,9 +209,9 @@ def _size_files(files: list[Path]) -> list[tuple[Path, int]]:
def _scan_file(root: Root, file_path: Path) -> IndexEntry | None:
"""Compute IndexEntry for a file. Blocking — run in executor.
- Files <= 40 MB are hashed in full (hash_version 1). Files > 40 MB use a
- 45 MB partial read — first 20 MB, last 20 MB, 5 MB at 50% — for
- hash_version 2."""
+ Files <= 9 MB are hashed in full (hash_version 1). Larger files hash their
+ size, first 4 MB, last 4 MB and 1 MB at 50% (hash_version 3). Version 2,
+ the earlier 45 MB sample, is still served from the cache, never computed."""
if not _is_indexable(file_path):
return None
try:
@@ -215,7 +220,7 @@ def _scan_file(root: Root, file_path: Path) -> IndexEntry | None:
return None
if stat.st_size > _PARTIAL_THRESHOLD:
hex_hash = _partial_hash(file_path, stat.st_size)
- hv = 2
+ hv = _PARTIAL_VERSION
else:
hasher = blake3.blake3()
with open(long_path(file_path), "rb") as f:
@@ -518,8 +523,12 @@ class DirectoryIndexer:
async def _hash_or_cached(self, root: Root, file_path: Path) -> IndexEntry | None:
"""
Cache-aware replacement for a bare _scan_file() call: skips the
- content read entirely when this path's (size, mtime, hash_version)
- still match what was hashed last time.
+ content read entirely when this path's (size, mtime) still match what
+ was hashed last time.
+
+ Whatever hash_version the cache holds is kept: TMDB matches, manual
+ corrections, thumbnails and members' playlists are keyed by the id, so
+ a file the node already knows never changes id because the scheme did.
"""
if not _is_indexable(file_path):
return None
@@ -530,11 +539,9 @@ class DirectoryIndexer:
if not _is_indexable_size(file_path, st.st_size):
return None
- expected_hv = 2 if st.st_size > _PARTIAL_THRESHOLD else 1
-
if self._cache is not None:
cached = await self._cache.lookup(
- str(file_path), st.st_size, st.st_mtime, expected_hv)
+ str(file_path), st.st_size, st.st_mtime)
if cached is not None:
return await self._attribute(IndexEntry(
id=cached.hash,