diff options
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py')
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py | 61 |
1 files changed, 48 insertions, 13 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py index a15da6f..a93df51 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py @@ -235,7 +235,9 @@ def _read_format_specific_tags(path: Path) -> tuple[dict, float | None] | None: return None -def _read_generic_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: +def _read_generic_tags_and_cover( + path: Path, skip_cover: bool = False, +) -> tuple[dict, float | None, bytes | None]: tags: dict = {} duration: float | None = None try: @@ -257,20 +259,23 @@ def _read_generic_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes tags["track_no"] = int(m.group()) cover: bytes | None = None - try: - raw = MutagenFile(str(path)) - except Exception: - raw = None - if raw is not None: + if not skip_cover: try: - cover = _extract_cover(raw) + raw = MutagenFile(str(path)) except Exception: - cover = None + raw = None + if raw is not None: + try: + cover = _extract_cover(raw) + except Exception: + cover = None return tags, duration, cover -def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: +def _read_tags_and_cover( + path: Path, skip_cover: bool = False, +) -> tuple[dict, float | None, bytes | None]: """ Synchronous — always called via asyncio.to_thread. Returns a partial `tags` dict (only keys actually found and not a known placeholder: @@ -278,13 +283,19 @@ def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: and raw cover bytes (None if absent, embedded and sibling-file both checked). Never raises for an unreadable/corrupt file — the caller falls back to filename parsing entirely in that case. + + `skip_cover` is set when the caller already has a cached cover for this + exact content (AudioEnricher._run, keyed by the file's own content + hash) — it skips the second file open plus the sibling-directory scan + entirely, neither of which the rename-refresh path needs redone: a + cover is derived from content, never from the filename. """ format_specific = _read_format_specific_tags(path) if format_specific is not None: tags, duration = format_specific cover = None # neither WMA nor Musepack has a convenient embedded-cover path here else: - tags, duration, cover = _read_generic_tags_and_cover(path) + tags, duration, cover = _read_generic_tags_and_cover(path, skip_cover=skip_cover) if "title" in tags: # Some taggers copy the bare filename into `title` verbatim, track @@ -292,7 +303,7 @@ def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: # that pollution would otherwise beat a cleaner one (docstring above). tags["title"] = title_parse.strip_track_prefix(tags["title"]) or tags["title"] - if cover is None: + if cover is None and not skip_cover: sibling = _find_sibling_cover(path.parent) if sibling is not None: try: @@ -422,9 +433,31 @@ class AudioEnricher: ) -> None: async with self._sem: fields: dict = {} + + # entry.id is the file's own content hash — a cover already + # cached under it means this exact content's cover was already + # extracted (this run, an earlier one, even a previous daemon + # process), so the second file open (embedded APIC/covr scan) + # and the sibling-directory disk read are both skippable. + # + # Tags themselves are *not* skipped this way, deliberately: + # unlike a cover, artist/album/title/track_no can fall back to + # the filename or the folder name (_artist_album_from_ancestors + # above) when no tag is present, which is exactly what a rename + # needs re-derived — _reenrich_renamed_audio_entries exists for + # precisely that. Caching the *result* the same way enrich.py's + # duration/width/height is cached doesn't apply cleanly here: + # mutagen reads tags and duration in the same call as cover, so + # skipping that call to save time would also skip the + # rename-sensitive fields, and skipping only the parts that are + # safe to skip needs the cover check below, not a separate + # cache of the tag-derived fields. + cached_thumb_hash = await self._media_cache.get_thumb_hash_by_file_id(entry.id) try: tags, duration, cover = await asyncio.wait_for( - asyncio.to_thread(_read_tags_and_cover, file_path), timeout=READ_TIMEOUT_SECS) + asyncio.to_thread( + _read_tags_and_cover, file_path, skip_cover=bool(cached_thumb_hash)), + timeout=READ_TIMEOUT_SECS) except Exception as e: log.warning("Tag read failed for %s: %s", file_path, e) tags, duration, cover = {}, None, None @@ -446,7 +479,9 @@ class AudioEnricher: fields["artist"] = artist fields["album"] = album - if cover: + if cached_thumb_hash: + fields["thumb_hash"] = cached_thumb_hash + elif cover: thumb_hash = blake3.blake3(cover).hexdigest() await self._media_cache.put_thumb(thumb_hash, entry.id, cover) fields["thumb_hash"] = thumb_hash |