""" Index-time enrichment for the Music group app: embedded tag/cover extraction (mutagen) and filename-parse fallback for a newly-added audio IndexEntry (docs/musicbay.md §2.1, §6). Runs through its own small bounded worker pool, the same discipline as the Videos app's `enrich.py` — separate from any other pool, never blocking a scan or the watchdog. Unlike `enrich.py`, this one shells out to nothing: `mutagen` is pure Python, synchronous I/O only, so there is no subprocess to spawn, no pipe to drain, and no ffmpeg-shaped deadlock risk here at all — the bounded pool exists to keep a large library's indexing burst bounded, not to contain a process. Reads run via `asyncio.to_thread` so they never block the event loop. MusicBrainz lookups are **not** done here. Tag/cover extraction is free and local, so it runs for every audio file the Music app is enabled for, regardless of whether MusicBrainz itself is turned on for the group — the flat view (docs/musicbay.md §5.2) needs nothing more than this. MusicBrainz is a separate, lazy, per-request enrichment (`music_meta_req`, handled in webrtc_server.py), the same "fetched on demand, cached once" shape TMDB already uses. **Revised 2026-08-24** against a real ~5700-file library (folder-per-artist mostly, but not uniformly — see musicbay.md's own "what got measured" note if one gets added). Two findings drove this revision, both confirmed with real data before writing the fix: 1. The original ancestor walk always went up two levels (parent = album, grandparent = artist) with no idea where the group's shared root itself was. For any file sitting in a *top-level* folder — common here: bare `Artist/track.mp3`, no album layer at all — the "grandparent" it found was the root directory's own name, so every such artist got relabelled as an album *of* a fake band named after the share. Measured: 289 of 5664 tracks (41 real, unrelated artists) collapsed into one bucket this way — the single biggest "artist" in the whole library, by a wide margin, which is exactly the kind of thing a person notices first and loses trust in. Fixed by passing the root's own path down so the walk refuses to use it as a name (`_artist_album_from_ancestors` below). 2. A tag being *present* is not the same as it being *meaningful*: some taggers write a literal placeholder ("No Artist", a French tool's "Nouvel artiste (334)") instead of leaving the field empty, and that placeholder is just as truthy as a real name — it was winning over the filename/folder fallback that would have done better. `_clean_tag` below turns a known placeholder back into "absent" before anything else sees it. WMA and Musepack read their tags outside mutagen's generic "easy" interface (neither format has an Easy* wrapper — `MutagenFile(path, easy=True)` just hands back the raw tag object, whose keys are format-specific and don't match the generic title/artist/album/tracknumber names the easy interface normally exposes elsewhere). Confirmed against real files before writing the fix: WMA's real keys are `Title`/`Author`/`WM/AlbumTitle`/ `WM/TrackNumber`, not "artist"/"album"; Musepack's are plain APEv2 keys (`Title`/`Artist`/`Album`/`Track`). Musepack also can't rely on mutagen's format auto-detection (`MutagenFile(path)` with no explicit class) — it misidentified a real .mpc file as MP3 often enough in a spot-check that the extension is used to pick the class directly instead. Neither format has a convenient embedded-cover path here, so both fall through to the sibling image-file fallback for cover art, same as everything else. """ import asyncio import logging import re from collections.abc import Awaitable, Callable from pathlib import Path import blake3 from meshbay_common.protocol import IndexEntry from mutagen import File as MutagenFile from mutagen.asf import ASF from mutagen.musepack import Musepack from meshbay_node.indexer import title_parse from meshbay_node.media_cache import MediaCache log = logging.getLogger(__name__) # Higher than the video pool's default (2): mutagen reads a few KB of tag # data synchronously, no subprocess, no decode — cheap enough that a wider # pool doesn't cost much and finishes a large library's initial scan sooner. DEFAULT_MAX_CONCURRENT = 4 READ_TIMEOUT_SECS = 10 _TRACK_NO_RE = re.compile(r"\d+") # Values some taggers write instead of leaving a field empty — treated as # "absent" so the fallback chain gets a chance instead of locking in a # placeholder. Not exhaustive by construction; each entry here was seen in # a real file, not guessed. "Various Artists" is deliberately *not* here — # that one is a real, meaningful compilation credit worth keeping as its # own bucket, not a placeholder. "Album inconnu ()"/standalone # "Inconnu" is a French ripping tool's own auto-generated placeholder, seen # on real WMA files — same idea as "Unknown Artist", different language. _PLACEHOLDER_RE = re.compile( r"^(no\s*artist|unknown(\s+artist)?|inconnu(e)?|" r"album\s+inconnu\s*\([^)]*\)|" r"nouvel(le)?\s+artiste\s*\(\d+\)|" r"nouveau\s+titre\s*\(\d+\)|track\s*\d*|)$", re.IGNORECASE, ) def _clean_tag(value: str | None) -> str | None: if not value: return None v = value.strip() return v if v and not _PLACEHOLDER_RE.match(v) else None def _extract_cover(mf) -> bytes | None: """ Best-effort embedded cover art across the tag formats mutagen exposes differently: ID3 (MP3) keeps pictures as APIC frames on `.tags`, FLAC exposes `.pictures` on the file object itself, MP4/M4A keeps a `covr` atom on `.tags`. Returns the first picture found, or None — most of a real library has no embedded art at all, which is not an error. """ tags = mf.tags if tags is not None and hasattr(tags, "getall"): pics = tags.getall("APIC") if pics: return bytes(pics[0].data) pictures = getattr(mf, "pictures", None) if pictures: return bytes(pictures[0].data) if tags is not None and hasattr(tags, "get"): covr = tags.get("covr") if covr: return bytes(covr[0]) return None # Loose image files sitting beside the tracks — very common for exactly the # era of rip this library mostly is (Windows Media Player's own # `Folder.jpg`/`AlbumArt_{guid}_(Large|Small).jpg`, or a manually dropped # `cover`/`front.jpg`). Measured: 267 such images across one real library, # none of them ever considered before this revision — embedded-art-only # missed them entirely. Ranked so a deliberately-named cover file wins over # WMP's cache thumbnail when both exist in the same folder. _COVER_STEM_RANK = ("cover", "folder", "front", "albumart") _COVER_EXTS = (".jpg", ".jpeg", ".png") def _find_sibling_cover(folder: Path) -> Path | None: try: candidates = [p for p in folder.iterdir() if p.is_file() and p.suffix.lower() in _COVER_EXTS] except OSError: return None if not candidates: return None def rank(p: Path) -> int: stem = p.stem.lower() for i, name in enumerate(_COVER_STEM_RANK): if name in stem: return i return len(_COVER_STEM_RANK) candidates.sort(key=lambda p: (rank(p), p.name)) return candidates[0] def _first_tag_value(values) -> str | None: return str(values[0]) if values else None def _asf_tags_from_object(asf_file: ASF) -> tuple[dict, float | None]: """ Pure mapping from an already-opened ASF object into our normalized dict — split out from the file-open call so the mapping itself can be exercised directly. Real ASF key names, confirmed against a real library sample and a freshly-encoded fixture alike: `Title`, `Author` (not "Artist"), `WM/AlbumTitle`, `WM/TrackNumber`. """ tags: dict = {} duration = getattr(asf_file.info, "length", None) if asf_file.info is not None else None if asf_file.tags is not None: title = _clean_tag(_first_tag_value(asf_file.tags.get("Title"))) if title: tags["title"] = title artist = _clean_tag(_first_tag_value(asf_file.tags.get("Author"))) if artist: tags["artist"] = artist album = _clean_tag(_first_tag_value(asf_file.tags.get("WM/AlbumTitle"))) if album: tags["album"] = album track_raw = _first_tag_value(asf_file.tags.get("WM/TrackNumber")) if track_raw: m = _TRACK_NO_RE.match(track_raw) if m: tags["track_no"] = int(m.group()) return tags, duration def _musepack_tags_from_object(mpc_file: Musepack) -> tuple[dict, float | None]: """ Pure mapping from an already-opened Musepack object's APEv2 tags into our normalized dict — split out the same way as `_asf_tags_from_object` (and for the same testability reason: there is no encoder available here to produce a real .mpc fixture from tags alone, mutagen included). """ tags: dict = {} duration = getattr(mpc_file.info, "length", None) if mpc_file.info is not None else None if mpc_file.tags is not None: for field, key in (("title", "Title"), ("artist", "Artist"), ("album", "Album")): cleaned = _clean_tag(_first_tag_value(mpc_file.tags.get(key))) if cleaned: tags[field] = cleaned track_raw = _first_tag_value(mpc_file.tags.get("Track")) if track_raw: m = _TRACK_NO_RE.match(track_raw) if m: tags["track_no"] = int(m.group()) return tags, duration def _read_format_specific_tags(path: Path) -> tuple[dict, float | None] | None: """ Returns None when the generic "easy" interface below already covers this extension — only WMA and Musepack need a bypass (docstring at the top of this module). """ suffix = path.suffix.lower() if suffix == ".wma": try: return _asf_tags_from_object(ASF(str(path))) except Exception: return {}, None if suffix == ".mpc": try: return _musepack_tags_from_object(Musepack(str(path))) except Exception: return {}, None return None def _read_generic_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: tags: dict = {} duration: float | None = None try: easy = MutagenFile(str(path), easy=True) except Exception: easy = None if easy is not None: if easy.info is not None: duration = getattr(easy.info, "length", None) for field in ("title", "artist", "album"): values = easy.get(field) cleaned = _clean_tag(str(values[0])) if values else None if cleaned: tags[field] = cleaned track_raw = easy.get("tracknumber") if track_raw: m = _TRACK_NO_RE.match(str(track_raw[0])) if m: tags["track_no"] = int(m.group()) cover: bytes | None = None try: raw = MutagenFile(str(path)) except Exception: raw = None if raw is not None: try: cover = _extract_cover(raw) except Exception: cover = None return tags, duration, cover def _read_tags_and_cover(path: Path) -> tuple[dict, float | None, bytes | None]: """ Synchronous — always called via asyncio.to_thread. Returns a partial `tags` dict (only keys actually found and not a known placeholder: title/artist/album/track_no), duration in seconds (None if unreadable), and raw cover bytes (None if absent, embedded and sibling-file both checked). Never raises for an unreadable/corrupt file — the caller falls back to filename parsing entirely in that case. """ format_specific = _read_format_specific_tags(path) if format_specific is not None: tags, duration = format_specific cover = None # neither WMA nor Musepack has a convenient embedded-cover path here else: tags, duration, cover = _read_generic_tags_and_cover(path) if "title" in tags: # Some taggers copy the bare filename into `title` verbatim, track # number included — a tag normally wins over the filename parse, so # that pollution would otherwise beat a cleaner one (docstring above). tags["title"] = title_parse.strip_track_prefix(tags["title"]) or tags["title"] if cover is None: sibling = _find_sibling_cover(path.parent) if sibling is not None: try: cover = sibling.read_bytes() except OSError: cover = None return tags, duration, cover # A folder name used as a last-resort artist/album, cleaned of the # punctuation-as-separator and release-tag noise this era of rip is full of # (underscores standing in for spaces, a bitrate/quality tag still attached # to the name). Same spirit as title_parse.naive_title for video, kept # separate because the junk vocabulary differs (bitrates and rip tags, not # edition/language tags). # A whole bracketed/parenthesized group is dropped if it contains any rip-tag # word ("[MP3 320kbps Album]", "(VBR HQ mp3s)") — matching one keyword at a # time left the rest of a multi-word group behind ("Album]" survived a first # version of this that only recognized "full album" as one fixed phrase). # Bare tokens outside brackets are stripped on their own. _RIP_TAG_GROUP_RE = re.compile( r"[\[\(][^\[\]\(\)]*\b(vbr|cbr|flac|mp3s?|kbps|hq|album)\b[^\[\]\(\)]*[\]\)]", re.IGNORECASE, ) _RIP_TAG_BARE_RE = re.compile(r"\b(vbr|cbr|flac|hq)\b", re.IGNORECASE) def _clean_folder_name(name: str) -> str: cleaned = re.sub(r"[._]+", " ", name) cleaned = _RIP_TAG_GROUP_RE.sub(" ", cleaned) cleaned = _RIP_TAG_BARE_RE.sub(" ", cleaned) cleaned = re.sub(r"\s+", " ", cleaned).strip(" -") return cleaned or name def _split_top_level_folder(name: str) -> tuple[str, str | None]: """ The one ancestor level left once a file's folder turns out to sit directly under the group's root (§ below) — there is no further ancestor to call "artist" without leaving the root entirely. The common convention for a single-release folder at that level is "Artist - Album ...junk..." (e.g. a soundtrack folder named after the film, or a release folder with the ripper's tags still attached); split on the first " - " when the cleaned name has one. Otherwise the whole (cleaned) name becomes the artist alone, which is the *more* common real shape here — a flat per-artist folder with no album subfolder at all. """ cleaned = _clean_folder_name(name) m = re.match(r"^(.{2,60}?)\s*-\s*(.{2,80})$", cleaned) if m: return m.group(1).strip(), m.group(2).strip() return cleaned, None def _artist_album_from_ancestors( file_path: Path, root_path: Path | None, ) -> tuple[str | None, str | None]: """ `Artist/Album/track.mp3` is one real shape in this library, but not the only one — measured directly against it (module docstring). This walk refuses to climb past `root_path` (the group's own shared directory): doing so used to read the root's own name as "the artist", which is never true and was the single largest source of bad groupings found. Three cases, by how many folders separate the file from the root: 0 (file sits directly in the root) — no folder context at all. 1 (a top-level folder) — ambiguous by construction; see `_split_top_level_folder`. 2+ — the classic Artist/Album layout: immediate parent is the album, its parent is the artist. `root_path` is None when the caller couldn't resolve which named root this entry belongs to (should not happen in practice — `entry.path` always names one — but the walk still terminates safely at the filesystem root rather than looping, same as before this revision). """ leaf = file_path.parent if root_path is not None and leaf == root_path: return None, None parent = leaf.parent if root_path is not None and parent == root_path: return _split_top_level_folder(leaf.name) if leaf == parent: # filesystem root reached without ever matching root_path return None, None return _clean_folder_name(parent.name), _clean_folder_name(leaf.name) class AudioEnricher: """Owns the node's bounded audio index-time enrichment pool.""" def __init__(self, media_cache: MediaCache, max_concurrent: int = DEFAULT_MAX_CONCURRENT): self._media_cache = media_cache self._sem = asyncio.Semaphore(max_concurrent) self._tasks: set[asyncio.Task] = set() def spawn( self, entry: IndexEntry, file_path: Path, on_done: Callable[[str, dict], Awaitable[None]], root_path: Path | None = None, ) -> asyncio.Task: """ Fire-and-forget one file's enrichment — same contract as `enrich.Enricher.spawn`: `on_done(file_id, fields)` is awaited with the index fields to merge once ready, never blocks the caller, and the returned task must be held by the caller for the same reason `WebRTCPeerSession._spawn` holds streaming tasks (a bare `ensure_future` can be garbage-collected mid-flight). `root_path` is the caller's own root boundary for this entry (daemon.py resolves it via `RootSet.split`) — see `_artist_album_from_ancestors`. """ task = asyncio.ensure_future(self._run(entry, file_path, on_done, root_path)) self._tasks.add(task) def _cleanup(t: asyncio.Task) -> None: self._tasks.discard(t) if not t.cancelled() and t.exception(): log.error("Audio enrichment failed for %s: %s", entry.id[:12], t.exception(), exc_info=t.exception()) task.add_done_callback(_cleanup) return task async def _run( self, entry: IndexEntry, file_path: Path, on_done: Callable[[str, dict], Awaitable[None]], root_path: Path | None, ) -> None: async with self._sem: fields: dict = {} try: tags, duration, cover = await asyncio.wait_for( asyncio.to_thread(_read_tags_and_cover, file_path), timeout=READ_TIMEOUT_SECS) except Exception as e: log.warning("Tag read failed for %s: %s", file_path, e) tags, duration, cover = {}, None, None if duration: fields["duration"] = int(duration) parsed = title_parse.parse_track_filename(entry.name) fields["display_title"] = tags.get("title") or parsed.title or parsed.naive_title fields["track_no"] = tags.get("track_no") if "track_no" in tags else parsed.track_no artist = tags.get("artist") album = tags.get("album") if not artist or not album: fallback_artist, fallback_album = await asyncio.to_thread( _artist_album_from_ancestors, file_path, root_path) artist = artist or fallback_artist album = album or fallback_album fields["artist"] = artist fields["album"] = album if cover: thumb_hash = blake3.blake3(cover).hexdigest() await self._media_cache.put_thumb(thumb_hash, entry.id, cover) fields["thumb_hash"] = thumb_hash await on_done(entry.id, fields)