diff options
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node')
4 files changed, 196 insertions, 34 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index 4dddc27..ae3f3dc 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -129,15 +129,40 @@ class Enricher: ) -> None: async with self._sem: fields: dict = {} + + # entry.id is the file's own content hash — a probe/thumbnail + # cache hit here means this exact content was already handled + # (this run, an earlier one, even a previous daemon process). + # Neither survives a restart on its own (the in-memory + # GroupIndex entry is rebuilt from scratch every time), but + # media_cache.db does — nothing was checking it before spawning + # ffprobe/ffmpeg again on every file, every restart. + # + # Deliberately *not* cached this way: display_title/season/ + # episode. Those come from guessit against entry.name, which is + # exactly what a rename needs re-derived — + # _reenrich_renamed_video_entries exists for precisely that — + # and reusing a stale parse under a new name would silently + # defeat it. ffprobe's own output has no such concern: the same + # bytes probe the same regardless of what the file is called. + cached_meta = await self._media_cache.get_video_meta(entry.id) duration: float | None = None - try: - _codec, duration, _has_audio, width, height, _raw = await asyncio.wait_for( - probe_video(str(file_path)), timeout=PROBE_TIMEOUT_SECS) - fields["duration"] = int(duration) if duration else None - fields["width"] = width - fields["height"] = height - except Exception as e: - log.warning("Probe failed for %s: %s", file_path, e) + if cached_meta is not None: + fields["duration"] = cached_meta["duration"] + fields["width"] = cached_meta["width"] + fields["height"] = cached_meta["height"] + duration = cached_meta["duration"] + else: + try: + _codec, duration, _has_audio, width, height, _raw = await asyncio.wait_for( + probe_video(str(file_path)), timeout=PROBE_TIMEOUT_SECS) + fields["duration"] = int(duration) if duration else None + fields["width"] = width + fields["height"] = height + except Exception as e: + log.warning("Probe failed for %s: %s", file_path, e) + await self._media_cache.put_video_meta( + entry.id, fields.get("duration"), fields.get("width"), fields.get("height")) ep = title_parse.parse_episode_filename(entry.name) if ep.episode is not None: @@ -153,14 +178,18 @@ class Enricher: mv = title_parse.parse_movie_filename(entry.name) fields["display_title"] = mv.display_title or mv.naive_title - try: - thumb = await _make_thumbnail(file_path, duration) - except Exception as e: - log.warning("Thumbnail generation failed for %s: %s", file_path, e) - thumb = None - if thumb: - thumb_hash = blake3.blake3(thumb).hexdigest() - await self._media_cache.put_thumb(thumb_hash, entry.id, thumb) - fields["thumb_hash"] = thumb_hash + cached_thumb_hash = await self._media_cache.get_thumb_hash_by_file_id(entry.id) + if cached_thumb_hash: + fields["thumb_hash"] = cached_thumb_hash + else: + try: + thumb = await _make_thumbnail(file_path, duration) + except Exception as e: + log.warning("Thumbnail generation failed for %s: %s", file_path, e) + thumb = None + if thumb: + thumb_hash = blake3.blake3(thumb).hexdigest() + await self._media_cache.put_thumb(thumb_hash, entry.id, thumb) + fields["thumb_hash"] = thumb_hash await on_done(entry.id, fields) diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py index a15da6f..a93df51 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py @@ -235,7 +235,9 @@ def _read_format_specific_tags(path: Path) -> tuple[dict, float | None] | None: return None -def _read_generic_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: +def _read_generic_tags_and_cover( + path: Path, skip_cover: bool = False, +) -> tuple[dict, float | None, bytes | None]: tags: dict = {} duration: float | None = None try: @@ -257,20 +259,23 @@ def _read_generic_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes tags["track_no"] = int(m.group()) cover: bytes | None = None - try: - raw = MutagenFile(str(path)) - except Exception: - raw = None - if raw is not None: + if not skip_cover: try: - cover = _extract_cover(raw) + raw = MutagenFile(str(path)) except Exception: - cover = None + raw = None + if raw is not None: + try: + cover = _extract_cover(raw) + except Exception: + cover = None return tags, duration, cover -def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: +def _read_tags_and_cover( + path: Path, skip_cover: bool = False, +) -> tuple[dict, float | None, bytes | None]: """ Synchronous — always called via asyncio.to_thread. Returns a partial `tags` dict (only keys actually found and not a known placeholder: @@ -278,13 +283,19 @@ def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: and raw cover bytes (None if absent, embedded and sibling-file both checked). Never raises for an unreadable/corrupt file — the caller falls back to filename parsing entirely in that case. + + `skip_cover` is set when the caller already has a cached cover for this + exact content (AudioEnricher._run, keyed by the file's own content + hash) — it skips the second file open plus the sibling-directory scan + entirely, neither of which the rename-refresh path needs redone: a + cover is derived from content, never from the filename. """ format_specific = _read_format_specific_tags(path) if format_specific is not None: tags, duration = format_specific cover = None # neither WMA nor Musepack has a convenient embedded-cover path here else: - tags, duration, cover = _read_generic_tags_and_cover(path) + tags, duration, cover = _read_generic_tags_and_cover(path, skip_cover=skip_cover) if "title" in tags: # Some taggers copy the bare filename into `title` verbatim, track @@ -292,7 +303,7 @@ def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: # that pollution would otherwise beat a cleaner one (docstring above). tags["title"] = title_parse.strip_track_prefix(tags["title"]) or tags["title"] - if cover is None: + if cover is None and not skip_cover: sibling = _find_sibling_cover(path.parent) if sibling is not None: try: @@ -422,9 +433,31 @@ class AudioEnricher: ) -> None: async with self._sem: fields: dict = {} + + # entry.id is the file's own content hash — a cover already + # cached under it means this exact content's cover was already + # extracted (this run, an earlier one, even a previous daemon + # process), so the second file open (embedded APIC/covr scan) + # and the sibling-directory disk read are both skippable. + # + # Tags themselves are *not* skipped this way, deliberately: + # unlike a cover, artist/album/title/track_no can fall back to + # the filename or the folder name (_artist_album_from_ancestors + # above) when no tag is present, which is exactly what a rename + # needs re-derived — _reenrich_renamed_audio_entries exists for + # precisely that. Caching the *result* the same way enrich.py's + # duration/width/height is cached doesn't apply cleanly here: + # mutagen reads tags and duration in the same call as cover, so + # skipping that call to save time would also skip the + # rename-sensitive fields, and skipping only the parts that are + # safe to skip needs the cover check below, not a separate + # cache of the tag-derived fields. + cached_thumb_hash = await self._media_cache.get_thumb_hash_by_file_id(entry.id) try: tags, duration, cover = await asyncio.wait_for( - asyncio.to_thread(_read_tags_and_cover, file_path), timeout=READ_TIMEOUT_SECS) + asyncio.to_thread( + _read_tags_and_cover, file_path, skip_cover=bool(cached_thumb_hash)), + timeout=READ_TIMEOUT_SECS) except Exception as e: log.warning("Tag read failed for %s: %s", file_path, e) tags, duration, cover = {}, None, None @@ -446,7 +479,9 @@ class AudioEnricher: fields["artist"] = artist fields["album"] = album - if cover: + if cached_thumb_hash: + fields["thumb_hash"] = cached_thumb_hash + elif cover: thumb_hash = blake3.blake3(cover).hexdigest() await self._media_cache.put_thumb(thumb_hash, entry.id, cover) fields["thumb_hash"] = thumb_hash diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich_photo.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich_photo.py index 44f9ebb..93a68cc 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich_photo.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich_photo.py @@ -143,6 +143,22 @@ class PhotoEnricher: on_done: Callable[[str, dict], Awaitable[None]], ) -> None: async with self._sem: + # entry.id is the file's own content hash — the same bytes + # produce the same thumbnail, so a hit here means this exact + # content was already decoded, resized and EXIF-read at some + # point (this run, an earlier one, even a previous daemon + # process — media_cache.db is the durable half of this). + # Without this check, every restart re-ran Pillow over every + # image in every configured photo_root from scratch — the + # in-memory GroupIndex enrichment fields don't survive a + # restart, but this cache does, and nothing was reading it + # before spawning the expensive work. + cached_hash = await self._media_cache.get_thumb_hash_by_file_id(entry.id) + cached_meta = await self._media_cache.get_photo_meta(entry.id) if cached_hash else None + if cached_hash and cached_meta: + await on_done(entry.id, {**cached_meta, "thumb_hash": cached_hash}) + return + fields: dict = {} try: thumb, width, height, taken_at, camera = await asyncio.wait_for( @@ -153,6 +169,7 @@ class PhotoEnricher: fields["camera"] = camera thumb_hash = blake3.blake3(thumb).hexdigest() await self._media_cache.put_thumb(thumb_hash, entry.id, thumb) + await self._media_cache.put_photo_meta(entry.id, width, height, taken_at, camera) fields["thumb_hash"] = thumb_hash except Exception as e: log.warning("Photo enrichment failed for %s: %s", file_path, e) diff --git a/packages/meshbay-node/src/meshbay_node/media_cache.py b/packages/meshbay-node/src/meshbay_node/media_cache.py index 47032b8..63cda41 100644 --- a/packages/meshbay-node/src/meshbay_node/media_cache.py +++ b/packages/meshbay-node/src/meshbay_node/media_cache.py @@ -54,6 +54,24 @@ CREATE TABLE IF NOT EXISTS season_meta ( fetched_at REAL NOT NULL, PRIMARY KEY (tmdb_id, season) ); +-- Videos app: the ffprobe fields enrich.py derives alongside the +-- thumbnail — duration/width/height only, deliberately *not* +-- display_title/season/episode. Those come from guessit against the +-- filename, which is exactly what a rename needs re-derived +-- (_reenrich_renamed_video_entries exists for precisely that); caching +-- them here would silently defeat that mechanism by handing back the old +-- name's parse under the new name. ffprobe's own output has no such +-- concern — the same bytes probe the same regardless of what the file is +-- called — so only the content-only fields are safe to skip recomputing. +-- thumb_hash is not duplicated here either, for the same reason it isn't +-- in photo_meta — get_thumb_hash_by_file_id(file_id) already answers that, +-- and it too is content-only (a frame grab doesn't depend on the name). +CREATE TABLE IF NOT EXISTS video_meta ( + file_id TEXT PRIMARY KEY, + duration INTEGER, + width INTEGER, + height INTEGER +); CREATE TABLE IF NOT EXISTS file_mbid ( file_id TEXT PRIMARY KEY, mbid TEXT NOT NULL @@ -63,6 +81,21 @@ CREATE TABLE IF NOT EXISTS mbid_meta ( json TEXT NOT NULL, fetched_at REAL NOT NULL ); +-- Photos app (docs/photos.md): the technical/EXIF fields enrich_photo.py +-- reads alongside the thumbnail. Durable for the same reason `thumbs` is — +-- without this, only the thumbnail bytes survived a restart, and every +-- image was still fully re-decoded through Pillow just to re-derive +-- width/height/taken_at/camera, which get_thumb_hash_by_file_id's own +-- cache hit had already proven unnecessary. thumb_hash is not duplicated +-- here — get_thumb_hash_by_file_id(file_id) already answers that, and a +-- second copy would just be one more place for the two to drift. +CREATE TABLE IF NOT EXISTS photo_meta ( + file_id TEXT PRIMARY KEY, + width INTEGER, + height INTEGER, + taken_at INTEGER, + camera TEXT +); """ # TMDB overviews/ratings do drift; a file's own resolved tmdb_id does not @@ -231,17 +264,65 @@ class MediaCache: ) await self._db.commit() + # ── photo technical/EXIF fields (Photos app) ───────────────────────────── + + async def get_photo_meta(self, file_id: str) -> dict | None: + async with self._db.execute( + "SELECT width, height, taken_at, camera FROM photo_meta WHERE file_id = ?", + (file_id,), + ) as cur: + row = await cur.fetchone() + if not row: + return None + return {"width": row[0], "height": row[1], "taken_at": row[2], "camera": row[3]} + + async def put_photo_meta(self, file_id: str, width: int | None, height: int | None, + taken_at: int | None, camera: str | None) -> None: + await self._db.execute( + "INSERT OR REPLACE INTO photo_meta (file_id, width, height, taken_at, camera) " + "VALUES (?, ?, ?, ?, ?)", + (file_id, width, height, taken_at, camera), + ) + await self._db.commit() + + # ── video technical fields (Videos app) ────────────────────────────────── + # + # duration/width/height only — see the schema comment on why + # display_title/season/episode are deliberately not cached here. + + async def get_video_meta(self, file_id: str) -> dict | None: + async with self._db.execute( + "SELECT duration, width, height FROM video_meta WHERE file_id = ?", + (file_id,), + ) as cur: + row = await cur.fetchone() + if not row: + return None + return {"duration": row[0], "width": row[1], "height": row[2]} + + async def put_video_meta(self, file_id: str, duration: int | None, + width: int | None, height: int | None) -> None: + await self._db.execute( + "INSERT OR REPLACE INTO video_meta (file_id, duration, width, height) " + "VALUES (?, ?, ?, ?)", + (file_id, duration, width, height), + ) + await self._db.commit() + # ── pruning ─────────────────────────────────────────────────────────────── async def prune_file(self, file_id: str) -> None: """ Called when a file leaves the index (deletion, unshared root). Removes - its thumbnail and its file->tmdb/file->mbid mappings. `tmdb_meta`/ - `mbid_meta` rows are left alone — they're keyed by tmdb_id/mbid, not - file_id, and other files (other episodes of the same show, other - tracks of the same release) may still reference the same entry. + its thumbnail, its per-app technical fields, and its file->tmdb/ + file->mbid mappings. `tmdb_meta`/`mbid_meta` rows are left alone — + they're keyed by tmdb_id/mbid, not file_id, and other files (other + episodes of the same show, other tracks of the same release) may + still reference the same entry. """ await self._db.execute("DELETE FROM thumbs WHERE file_id = ?", (file_id,)) + await self._db.execute("DELETE FROM photo_meta WHERE file_id = ?", (file_id,)) + await self._db.execute("DELETE FROM video_meta WHERE file_id = ?", (file_id,)) await self._db.execute("DELETE FROM file_tmdb WHERE file_id = ?", (file_id,)) await self._db.execute("DELETE FROM file_mbid WHERE file_id = ?", (file_id,)) await self._db.commit() |