""" MeshBay Node — persistent (path, size, mtime) -> hash cache, one per group. Without this, every node restart re-reads and re-hashes every file in every root, even when nothing changed — measured at 23 minutes for a 114 GB library on a USB hard drive. This cache lets a scan skip the read entirely for a file whose size and mtime still match what was hashed last time. It is a path-keyed accelerator only. The GroupIndex itself stays keyed by content hash (see indexer.py's note on why two identical files are one entry) — this cache never changes that, it only avoids recomputing a hash that has not changed. """ import logging from dataclasses import dataclass from pathlib import Path import aiosqlite log = logging.getLogger(__name__) _SCHEMA = """ CREATE TABLE IF NOT EXISTS files ( path TEXT PRIMARY KEY, mtime REAL NOT NULL, size INTEGER NOT NULL, hash TEXT NOT NULL, type TEXT NOT NULL, added_at INTEGER NOT NULL ); """ @dataclass class CachedEntry: hash: str type: str added_at: int class IndexCache: """Async SQLite (path, size, mtime) -> hash cache for one group.""" def __init__(self, db_path: Path): self._db_path = db_path self._db: aiosqlite.Connection | None = None async def open(self) -> None: self._db_path.parent.mkdir(parents=True, exist_ok=True) self._db = await aiosqlite.connect(str(self._db_path)) await self._db.executescript(_SCHEMA) await self._db.commit() async def close(self) -> None: if self._db: await self._db.close() self._db = None async def __aenter__(self): await self.open() return self async def __aexit__(self, *_): await self.close() async def lookup(self, path: str, size: int, mtime: float) -> CachedEntry | None: """ A cache hit requires an EXACT match on both size and mtime. A mtime touched without a content change is a false negative (an unnecessary rehash) — accepted, since the alternative (trusting a stale hash) is a silent wrong answer instead of an occasional wasted read. """ async with self._db.execute( "SELECT hash, type, added_at FROM files " "WHERE path = ? AND size = ? AND mtime = ?", (path, size, mtime)) as cur: row = await cur.fetchone() return CachedEntry(hash=row[0], type=row[1], added_at=row[2]) if row else None async def put(self, path: str, size: int, mtime: float, hash: str, type: str, added_at: int) -> None: """ Written only once a file has been hashed in full — never partway through — so a crash mid-hash leaves no stale/partial row behind: the next scan simply finds no cache entry and hashes the file again. """ await self._db.execute( "INSERT INTO files (path, mtime, size, hash, type, added_at) " "VALUES (?, ?, ?, ?, ?, ?) " "ON CONFLICT(path) DO UPDATE SET " "mtime = excluded.mtime, size = excluded.size, hash = excluded.hash, " "type = excluded.type, added_at = excluded.added_at", (path, mtime, size, hash, type, added_at)) await self._db.commit()