diff options
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/indexer/indexer.py')
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/indexer/indexer.py | 33 |
1 files changed, 20 insertions, 13 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/indexer.py b/packages/meshbay-node/src/meshbay_node/indexer/indexer.py index f06d30f..4eeeec0 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/indexer.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/indexer.py @@ -97,10 +97,14 @@ def _is_indexable_size(path: Path, size: int) -> bool: _HASH_CHUNK = 8 * 1024 * 1024 # 8 MB streaming hash chunks -_PARTIAL_THRESHOLD = 40 * 1024 * 1024 # files above this use partial-read hashing -_PARTIAL_HEAD = 20 * 1024 * 1024 -_PARTIAL_TAIL = 20 * 1024 * 1024 -_PARTIAL_MID = 5 * 1024 * 1024 +# On a spinning disk the cost of a sample is its three seeks, not its bytes: +# measured cold on USB, 45 MB took 516 ms a file, 9 MB 94 ms, 3 MB 79 ms. The +# threshold is the sample size, so no file costs more to read than a sample. +_PARTIAL_THRESHOLD = 9 * 1024 * 1024 # files above this use partial-read hashing +_PARTIAL_HEAD = 4 * 1024 * 1024 +_PARTIAL_TAIL = 4 * 1024 * 1024 +_PARTIAL_MID = 1 * 1024 * 1024 +_PARTIAL_VERSION = 3 def _feed(hasher, f, nbytes: int) -> None: @@ -115,6 +119,7 @@ def _feed(hasher, f, nbytes: int) -> None: def _partial_hash(file_path: Path, size: int) -> str: hasher = blake3.blake3() + hasher.update(size.to_bytes(8, "little")) with open(long_path(file_path), "rb") as f: _feed(hasher, f, _PARTIAL_HEAD) f.seek(size - _PARTIAL_TAIL) @@ -204,9 +209,9 @@ def _size_files(files: list[Path]) -> list[tuple[Path, int]]: def _scan_file(root: Root, file_path: Path) -> IndexEntry | None: """Compute IndexEntry for a file. Blocking — run in executor. - Files <= 40 MB are hashed in full (hash_version 1). Files > 40 MB use a - 45 MB partial read — first 20 MB, last 20 MB, 5 MB at 50% — for - hash_version 2.""" + Files <= 9 MB are hashed in full (hash_version 1). Larger files hash their + size, first 4 MB, last 4 MB and 1 MB at 50% (hash_version 3). Version 2, + the earlier 45 MB sample, is still served from the cache, never computed.""" if not _is_indexable(file_path): return None try: @@ -215,7 +220,7 @@ def _scan_file(root: Root, file_path: Path) -> IndexEntry | None: return None if stat.st_size > _PARTIAL_THRESHOLD: hex_hash = _partial_hash(file_path, stat.st_size) - hv = 2 + hv = _PARTIAL_VERSION else: hasher = blake3.blake3() with open(long_path(file_path), "rb") as f: @@ -518,8 +523,12 @@ class DirectoryIndexer: async def _hash_or_cached(self, root: Root, file_path: Path) -> IndexEntry | None: """ Cache-aware replacement for a bare _scan_file() call: skips the - content read entirely when this path's (size, mtime, hash_version) - still match what was hashed last time. + content read entirely when this path's (size, mtime) still match what + was hashed last time. + + Whatever hash_version the cache holds is kept: TMDB matches, manual + corrections, thumbnails and members' playlists are keyed by the id, so + a file the node already knows never changes id because the scheme did. """ if not _is_indexable(file_path): return None @@ -530,11 +539,9 @@ class DirectoryIndexer: if not _is_indexable_size(file_path, st.st_size): return None - expected_hv = 2 if st.st_size > _PARTIAL_THRESHOLD else 1 - if self._cache is not None: cached = await self._cache.lookup( - str(file_path), st.st_size, st.st_mtime, expected_hv) + str(file_path), st.st_size, st.st_mtime) if cached is not None: return await self._attribute(IndexEntry( id=cached.hash, |