From 584730f9486e153c6c477293f03a95e78f5ab5f3 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Mon, 24 Aug 2026 18:30:00 +0200 Subject: fix(node): stop inventing a fake artist from the shared root's name MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Music grouping was measured against a real ~5700-file library and came back worse than a plain file listing. Root cause: the artist/album ancestor walk always climbed exactly two levels (parent = album, grandparent = artist) with no idea where the group's own shared root was. Any file in a flat top-level folder — common here: bare `Artist/track.mp3`, no album subfolder at all — had its "grandparent" resolve to the root directory's own name, so the artist got replaced by the share's name. Measured: 289 of 5664 tracks across 41 real, unrelated artists (Ben Harper, Dire Straits, Jimi Hendrix, Janis Joplin, ...) collapsed into one fake artist this way — the single biggest bucket in the whole library, ahead of every real one. - `_artist_album_from_ancestors` now takes the entry's own root boundary (daemon.py resolves it via `RootSet.split`) and refuses to read it as a name. A file sitting in a top-level folder — genuinely ambiguous, artist or a standalone album/compilation — is handled by `_split_top_level_folder`: split on "Artist - Album" when the (cleaned) folder name has that shape, otherwise the whole name becomes the artist alone, the more common real case here. - `_clean_tag` treats known tagger placeholders ("No Artist", a French tool's "Nouvel artiste (334)") as absent rather than a real value — they were just as truthy as a real name and were locking out the fallback that would have done better. "Various Artists" is kept, a real compilation credit rather than a placeholder. - A `title` tag that's the bare filename copied verbatim (track number included — found live on a whole CD-single) is stripped through the same prefix rule the filename parser already used (`title_parse.strip_track_prefix`), since a tag normally wins over the parsed title. - Cover art: only 11% of a 400-file sample had embedded art (expected for this era of rip), but 267 loose cover images sit beside the tracks across the library (Windows Media Player's `Folder.jpg`/ `AlbumArt_{guid}_*.jpg`, manual `cover.jpg`) and were never looked at. `_find_sibling_cover` checks the track's own folder before giving up — measured coverage 11% -> 26% on the same library, zero network calls. `_enriched_attempted` is in-memory and resets on restart, so a node restart is enough to re-run enrichment over an already-scanned library with the fixed logic — no rescan flag, no cache to clear by hand. 22 tests in test_enrich_audio.py (11 new), including the exact regression case end to end through the real pool. Full suite: 1129 passed, no regressions. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01KBi7ALLGfwcjBXt57yNMcy --- packages/meshbay-node/src/meshbay_node/daemon.py | 9 +- .../src/meshbay_node/indexer/enrich_audio.py | 197 ++++++++++++++++++--- .../src/meshbay_node/indexer/title_parse.py | 14 ++ 3 files changed, 196 insertions(+), 24 deletions(-) (limited to 'packages/meshbay-node/src/meshbay_node') diff --git a/packages/meshbay-node/src/meshbay_node/daemon.py b/packages/meshbay-node/src/meshbay_node/daemon.py index 913ad68..ae4e1f0 100644 --- a/packages/meshbay-node/src/meshbay_node/daemon.py +++ b/packages/meshbay-node/src/meshbay_node/daemon.py @@ -1168,11 +1168,18 @@ class NodeDaemon: if not file_path or not file_path.exists(): continue self._enriched_attempted.add(entry.id) + # The entry's own named root, so the ancestor walk + # (enrich_audio._artist_album_from_ancestors) can tell "a + # top-level folder under this root" from "one level deeper" — + # without this boundary it used to read the root's own + # directory name as an artist for every flat top-level folder. + split = indexer.roots.split(entry.path) + root_path = split[0].path if split else None async def on_done(file_id: str, fields: dict, _indexer=indexer) -> None: await self._on_enriched(_indexer, file_id, fields) - self._audio_enricher.spawn(entry, file_path, on_done) + self._audio_enricher.spawn(entry, file_path, on_done, root_path) async def _enrich_music_now(self, group_id: str) -> None: """ diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py index 17a58d6..9ecf1b9 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich_audio.py @@ -19,6 +19,30 @@ flat view (docs/musicbay.md §5.2) needs nothing more than this. MusicBrainz is a separate, lazy, per-request enrichment (`music_meta_req`, handled in webrtc_server.py), the same "fetched on demand, cached once" shape TMDB already uses. + +**Revised 2026-08-24** against a real ~5700-file library (folder-per-artist +mostly, but not uniformly — see musicbay.md's own "what got measured" note +if one gets added). Two findings drove this revision, both confirmed with +real data before writing the fix: + +1. The original ancestor walk always went up two levels (parent = album, + grandparent = artist) with no idea where the group's shared root itself + was. For any file sitting in a *top-level* folder — common here: bare + `Artist/track.mp3`, no album layer at all — the "grandparent" it found + was the root directory's own name, so every such artist got relabelled + as an album *of* a fake band named after the share. Measured: 289 of + 5664 tracks (41 real, unrelated artists) collapsed into one bucket this + way — the single biggest "artist" in the whole library, by a wide + margin, which is exactly the kind of thing a person notices first and + loses trust in. Fixed by passing the root's own path down so the walk + refuses to use it as a name (`_artist_album_from_ancestors` below). +2. A tag being *present* is not the same as it being *meaningful*: some + taggers write a literal placeholder ("No Artist", a French tool's + "Nouvel artiste (334)") instead of leaving the field empty, and that + placeholder is just as truthy as a real name — it was winning over the + filename/folder fallback that would have done better. `_clean_tag` + below turns a known placeholder back into "absent" before anything else + sees it. """ import asyncio @@ -44,6 +68,25 @@ READ_TIMEOUT_SECS = 10 _TRACK_NO_RE = re.compile(r"\d+") +# Values some taggers write instead of leaving a field empty — treated as +# "absent" so the fallback chain gets a chance instead of locking in a +# placeholder. Not exhaustive by construction; each entry here was seen in +# a real file, not guessed. "Various Artists" is deliberately *not* here — +# that one is a real, meaningful compilation credit worth keeping as its +# own bucket, not a placeholder. +_PLACEHOLDER_RE = re.compile( + r"^(no\s*artist|unknown(\s+artist)?|nouvel(le)?\s+artiste\s*\(\d+\)|" + r"nouveau\s+titre\s*\(\d+\)|track\s*\d*|)$", + re.IGNORECASE, +) + + +def _clean_tag(value: str | None) -> str | None: + if not value: + return None + v = value.strip() + return v if v and not _PLACEHOLDER_RE.match(v) else None + def _extract_cover(mf) -> bytes | None: """ @@ -68,13 +111,44 @@ def _extract_cover(mf) -> bytes | None: return None +# Loose image files sitting beside the tracks — very common for exactly the +# era of rip this library mostly is (Windows Media Player's own +# `Folder.jpg`/`AlbumArt_{guid}_(Large|Small).jpg`, or a manually dropped +# `cover`/`front.jpg`). Measured: 267 such images across one real library, +# none of them ever considered before this revision — embedded-art-only +# missed them entirely. Ranked so a deliberately-named cover file wins over +# WMP's cache thumbnail when both exist in the same folder. +_COVER_STEM_RANK = ("cover", "folder", "front", "albumart") +_COVER_EXTS = (".jpg", ".jpeg", ".png") + + +def _find_sibling_cover(folder: Path) -> Path | None: + try: + candidates = [p for p in folder.iterdir() + if p.is_file() and p.suffix.lower() in _COVER_EXTS] + except OSError: + return None + if not candidates: + return None + + def rank(p: Path) -> int: + stem = p.stem.lower() + for i, name in enumerate(_COVER_STEM_RANK): + if name in stem: + return i + return len(_COVER_STEM_RANK) + candidates.sort(key=lambda p: (rank(p), p.name)) + return candidates[0] + + def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: """ Synchronous — always called via asyncio.to_thread. Returns a partial - `tags` dict (only keys actually found: title/artist/album/track_no), - duration in seconds (None if unreadable), and raw cover bytes (None if - absent). Never raises for an unreadable/corrupt file — the caller falls - back to filename parsing entirely in that case. + `tags` dict (only keys actually found and not a known placeholder: + title/artist/album/track_no), duration in seconds (None if unreadable), + and raw cover bytes (None if absent, embedded and sibling-file both + checked). Never raises for an unreadable/corrupt file — the caller + falls back to filename parsing entirely in that case. """ tags: dict = {} duration: float | None = None @@ -87,13 +161,19 @@ def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: duration = getattr(easy.info, "length", None) for field in ("title", "artist", "album"): values = easy.get(field) - if values and str(values[0]).strip(): - tags[field] = str(values[0]).strip() + cleaned = _clean_tag(str(values[0])) if values else None + if cleaned: + tags[field] = cleaned track_raw = easy.get("tracknumber") if track_raw: m = _TRACK_NO_RE.match(str(track_raw[0])) if m: tags["track_no"] = int(m.group()) + if "title" in tags: + # Some taggers copy the bare filename into `title` verbatim, track + # number included — a tag normally wins over the filename parse, so + # that pollution would otherwise beat a cleaner one (docstring above). + tags["title"] = title_parse.strip_track_prefix(tags["title"]) or tags["title"] cover: bytes | None = None try: @@ -105,27 +185,94 @@ def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: cover = _extract_cover(raw) except Exception: cover = None + if cover is None: + sibling = _find_sibling_cover(path.parent) + if sibling is not None: + try: + cover = sibling.read_bytes() + except OSError: + cover = None return tags, duration, cover -def _artist_album_from_ancestors(file_path: Path) -> tuple[str | None, str | None]: +# A folder name used as a last-resort artist/album, cleaned of the +# punctuation-as-separator and release-tag noise this era of rip is full of +# ("L_Oeuf_Raide_-_Berlin_Eggsile", "Sinsemilia - Premiere Recolte [MP3 +# 320kbps Album]"). Same spirit as title_parse.naive_title for video, kept +# separate because the junk vocabulary differs (bitrates and rip tags, not +# edition/language tags). +# A whole bracketed/parenthesized group is dropped if it contains any rip-tag +# word ("[MP3 320kbps Album]", "(VBR HQ mp3s)") — matching one keyword at a +# time left the rest of a multi-word group behind ("Album]" survived a first +# version of this that only recognized "full album" as one fixed phrase). +# Bare tokens outside brackets are stripped on their own. +_RIP_TAG_GROUP_RE = re.compile( + r"[\[\(][^\[\]\(\)]*\b(vbr|cbr|flac|mp3s?|kbps|hq|album)\b[^\[\]\(\)]*[\]\)]", + re.IGNORECASE, +) +_RIP_TAG_BARE_RE = re.compile(r"\b(vbr|cbr|flac|hq)\b", re.IGNORECASE) + + +def _clean_folder_name(name: str) -> str: + cleaned = re.sub(r"[._]+", " ", name) + cleaned = _RIP_TAG_GROUP_RE.sub(" ", cleaned) + cleaned = _RIP_TAG_BARE_RE.sub(" ", cleaned) + cleaned = re.sub(r"\s+", " ", cleaned).strip(" -") + return cleaned or name + + +def _split_top_level_folder(name: str) -> tuple[str, str | None]: """ - `Artist/Album/track.mp3` is the common shape (docs/musicbay.md §2.1) — - used only to fill whatever the tags left empty. No attempt to validate - against the group's actual roots (enrich.py's ancestor walks don't - either): a flat `Artist/track.mp3` layout, or a various-artists - compilation folder, just yields a plausible-but-not-guaranteed album - name from the immediate parent and nothing further up — good enough for - a fallback, not asserted as accurate. + The one ancestor level left once a file's folder turns out to sit + directly under the group's root (§ below) — there is no further + ancestor to call "artist" without leaving the root entirely. The common + convention for a single-release folder at that level is + "Artist - Album ...junk..." ("GHOST DOG - Soundtrack", + "cypress_hill_-los_grandes__xitos_en_espa_ol"); split on the first + " - " when the cleaned name has one. Otherwise the whole (cleaned) name + becomes the artist alone, which is the *more* common real shape here — + a flat per-artist folder with no album subfolder at all ("Ben Harper", + "bob_marley", "Renaud"). """ - album_folder = file_path.parent - if album_folder == album_folder.parent: + cleaned = _clean_folder_name(name) + m = re.match(r"^(.{2,60}?)\s*-\s*(.{2,80})$", cleaned) + if m: + return m.group(1).strip(), m.group(2).strip() + return cleaned, None + + +def _artist_album_from_ancestors( + file_path: Path, root_path: Path | None, +) -> tuple[str | None, str | None]: + """ + `Artist/Album/track.mp3` is one real shape in this library, but not the + only one — measured directly against it (module docstring). This walk + refuses to climb past `root_path` (the group's own shared directory): + doing so used to read the root's own name as "the artist", which is + never true and was the single largest source of bad groupings found. + + Three cases, by how many folders separate the file from the root: + 0 (file sits directly in the root) — no folder context at all. + 1 (a top-level folder) — ambiguous by construction; see + `_split_top_level_folder`. + 2+ — the classic Artist/Album layout: immediate parent is the album, + its parent is the artist. + + `root_path` is None when the caller couldn't resolve which named root + this entry belongs to (should not happen in practice — `entry.path` + always names one — but the walk still terminates safely at the + filesystem root rather than looping, same as before this revision). + """ + leaf = file_path.parent + if root_path is not None and leaf == root_path: + return None, None + parent = leaf.parent + if root_path is not None and parent == root_path: + return _split_top_level_folder(leaf.name) + if leaf == parent: # filesystem root reached without ever matching root_path return None, None - artist_folder = album_folder.parent - album = album_folder.name or None - artist = artist_folder.name if artist_folder != artist_folder.parent else None - return artist, album + return _clean_folder_name(parent.name), _clean_folder_name(leaf.name) class AudioEnricher: @@ -139,6 +286,7 @@ class AudioEnricher: def spawn( self, entry: IndexEntry, file_path: Path, on_done: Callable[[str, dict], Awaitable[None]], + root_path: Path | None = None, ) -> asyncio.Task: """ Fire-and-forget one file's enrichment — same contract as @@ -146,9 +294,11 @@ class AudioEnricher: the index fields to merge once ready, never blocks the caller, and the returned task must be held by the caller for the same reason `WebRTCPeerSession._spawn` holds streaming tasks (a bare - `ensure_future` can be garbage-collected mid-flight). + `ensure_future` can be garbage-collected mid-flight). `root_path` is + the caller's own root boundary for this entry (daemon.py resolves + it via `RootSet.split`) — see `_artist_album_from_ancestors`. """ - task = asyncio.ensure_future(self._run(entry, file_path, on_done)) + task = asyncio.ensure_future(self._run(entry, file_path, on_done, root_path)) self._tasks.add(task) def _cleanup(t: asyncio.Task) -> None: @@ -162,6 +312,7 @@ class AudioEnricher: async def _run( self, entry: IndexEntry, file_path: Path, on_done: Callable[[str, dict], Awaitable[None]], + root_path: Path | None, ) -> None: async with self._sem: fields: dict = {} @@ -183,7 +334,7 @@ class AudioEnricher: album = tags.get("album") if not artist or not album: fallback_artist, fallback_album = await asyncio.to_thread( - _artist_album_from_ancestors, file_path) + _artist_album_from_ancestors, file_path, root_path) artist = artist or fallback_artist album = album or fallback_album fields["artist"] = artist diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 6676e9b..5203247 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -187,6 +187,20 @@ def parse_track_filename(filename: str) -> ParsedTrack: return ParsedTrack(title=title, track_no=track_no, naive_title=naive_title(filename)) +def strip_track_prefix(text: str) -> str: + """ + Some taggers copy the bare filename into the `title` tag verbatim, + track-number prefix included (found live: a whole CD-single's worth of + `title` tags reading "01 - Venus As A Boy" rather than "Venus As A + Boy") — since a tag normally wins over the filename-parsed title + (enrich_audio.py), that pollution would otherwise beat a cleaner parse. + A no-op when there's no such prefix, so a genuinely clean tag is + returned unchanged. + """ + m = _TRACK_PREFIX_RE.match(text) + return re.sub(r"^[\s-]+", "", text[m.end():]).strip() if m else text + + def parse_episode_filename(filename: str) -> ParsedName: """ Parse an episode filename. `display_title` may come back None (e.g. -- cgit v1.2.3