aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/tests/test_enrich_audio.py
blob: 1485a1f0e5144aa7105f3ae8ffc1c04e4d3aeaac (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
"""Tests for indexer/enrich_audio.py — tag/cover extraction and the end-to-end pool."""

import asyncio
import shutil
import subprocess
from pathlib import Path

import pytest
from meshbay_common.protocol import IndexEntry
from meshbay_node.indexer.enrich_audio import (
    AudioEnricher,
    _artist_album_from_ancestors,
    _clean_tag,
    _extract_cover,
    _musepack_tags_from_object,
    _split_top_level_folder,
)
from meshbay_node.indexer.title_parse import parse_track_filename, strip_track_prefix
from meshbay_node.media_cache import MediaCache

_HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe")


# ── pure helpers, no ffmpeg/mutagen file needed ──────────────────────────────

def test_parse_track_filename_splits_leading_track_number():
    parsed = parse_track_filename("01 - Venus As A Boy (Edited Lp Version).mp3")
    assert parsed.track_no == 1
    assert parsed.title == "Venus As A Boy (Edited Lp Version)"


def test_parse_track_filename_handles_dot_separated():
    parsed = parse_track_filename("03. Human Behaviour.mp3")
    assert parsed.track_no == 3
    assert parsed.title == "Human Behaviour"


def test_parse_track_filename_handles_underscore_separated():
    parsed = parse_track_filename("12_Some_Title.mp3")
    assert parsed.track_no == 12
    assert parsed.title == "Some Title"


def test_parse_track_filename_no_prefix_leaves_track_no_none():
    parsed = parse_track_filename("Some Title.mp3")
    assert parsed.track_no is None
    assert parsed.title == "Some Title"


def test_parse_track_filename_does_not_mistake_a_leading_year_for_a_track_number():
    parsed = parse_track_filename("1999 - Some Title.mp3")
    assert parsed.track_no is None, "a 4-digit prefix is capped out, not read as track 199"


def test_artist_album_from_ancestors_reads_artist_album_track_layout(tmp_path):
    folder = tmp_path / "Some Artist" / "Some Album"
    folder.mkdir(parents=True)
    track = folder / "01 - A Track.mp3"
    track.touch()

    artist, album = _artist_album_from_ancestors(track, tmp_path)

    assert artist == "Some Artist"
    assert album == "Some Album"


def test_artist_album_from_ancestors_refuses_to_name_the_root_as_artist(tmp_path):
    """
    The regression this whole revision exists for: a flat `Artist/track.mp3`
    layout (no album subfolder) used to read the *root's own directory
    name* as the artist, because the walk always climbed two levels with no
    idea where the root was. Measured live against a real library: dozens
    of unrelated artists collapsed into one fake "artist" this way — the
    single biggest bucket in the whole library.
    """
    folder = tmp_path / "Some Flat Artist"
    folder.mkdir(parents=True)
    track = folder / "A Track.mp3"
    track.touch()

    artist, album = _artist_album_from_ancestors(track, tmp_path)

    assert artist == "Some Flat Artist"
    assert album is None, "no album folder exists — must not invent one, or swap artist/album"


def test_artist_album_from_ancestors_file_directly_in_root_has_no_context(tmp_path):
    track = tmp_path / "loose_track.mp3"
    track.touch()

    artist, album = _artist_album_from_ancestors(track, tmp_path)

    assert (artist, album) == (None, None)


def test_artist_album_from_ancestors_without_a_root_keeps_the_old_two_level_behaviour(tmp_path):
    """No root resolved (should not happen in practice, but must degrade safely)."""
    folder = tmp_path / "Some Artist" / "Some Album"
    folder.mkdir(parents=True)
    track = folder / "01 - A Track.mp3"
    track.touch()

    artist, album = _artist_album_from_ancestors(track, None)

    assert artist == "Some Artist"
    assert album == "Some Album"


def test_split_top_level_folder_splits_artist_dash_album():
    artist, album = _split_top_level_folder("Some Movie - Soundtrack")
    assert artist == "Some Movie"
    assert album == "Soundtrack"


def test_split_top_level_folder_with_no_separator_is_artist_only():
    artist, album = _split_top_level_folder("Some Flat Artist")
    assert artist == "Some Flat Artist"
    assert album is None


def test_split_top_level_folder_cleans_rip_tag_noise():
    artist, album = _split_top_level_folder(
        "Some Artist - Some Release [MP3 320kbps Album]")
    assert artist == "Some Artist"
    assert album == "Some Release"


def test_clean_tag_filters_known_placeholders():
    assert _clean_tag("No Artist") is None
    assert _clean_tag("unknown artist") is None
    assert _clean_tag("Nouvel artiste (334)") is None
    assert _clean_tag("Nouveau titre (12)") is None
    assert _clean_tag("") is None
    assert _clean_tag(None) is None


def test_clean_tag_keeps_various_artists_as_a_real_credit():
    """A real, meaningful compilation credit — not a placeholder to blank out."""
    assert _clean_tag("Various Artists") == "Various Artists"


def test_clean_tag_filters_a_french_ripper_unknown_album_placeholder():
    """Seen on a real WMA file: a French tool's auto-generated placeholder,
    date stamp included — not a real album title."""
    assert _clean_tag("Album inconnu (20/11/2003 14:00:03)") is None
    assert _clean_tag("Inconnu") is None


def test_clean_tag_keeps_a_real_value():
    # Non-ASCII must not be mistaken for a placeholder pattern.
    assert _clean_tag("Ünïqùé Ärtïst") == "Ünïqùé Ärtïst"


def test_strip_track_prefix_removes_a_leaked_filename_number():
    assert strip_track_prefix("01 - Some Track (Edited Version)") == \
        "Some Track (Edited Version)"


def test_strip_track_prefix_is_a_noop_on_a_clean_title():
    assert strip_track_prefix("Some Track") == "Some Track"


# ── end-to-end against a real (tiny, synthetic) MP3 file ────────────────────

pytestmark_ffmpeg = pytest.mark.skipif(not _HAVE_FFMPEG, reason="ffmpeg/ffprobe not installed")


def _make_clip(path: Path, *, title=None, artist=None, album=None, track=None) -> None:
    subprocess.run(
        ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
         "-f", "lavfi", "-i", "sine=frequency=440:duration=1",
         "-c:a", "libmp3lame", "-b:a", "64k",
         *(["-metadata", f"title={title}"] if title else []),
         *(["-metadata", f"artist={artist}"] if artist else []),
         *(["-metadata", f"album={album}"] if album else []),
         *(["-metadata", f"track={track}"] if track else []),
         str(path)],
        check=True, capture_output=True,
    )


@pytest.fixture
async def media_cache(tmp_path):
    c = MediaCache(db_path=tmp_path / "media_cache.db")
    await c.open()
    yield c
    await c.close()


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_prefers_tags_over_filename_parse(tmp_path, media_cache):
    clip = tmp_path / "99 - wrong title.mp3"
    _make_clip(clip, title="Real Title", artist="Real Artist", album="Real Album", track=3)
    entry = IndexEntry(id="fileid1", name=clip.name, path=clip.name,
                        size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, clip, on_done)
    file_id, fields = await asyncio.wait_for(done, timeout=30)

    assert file_id == "fileid1"
    assert fields["display_title"] == "Real Title"
    assert fields["artist"] == "Real Artist"
    assert fields["album"] == "Real Album"
    assert fields["track_no"] == 3
    assert fields["duration"] == 1
    assert fields.get("thumb_hash") is None, "no embedded cover was written in this clip"


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_falls_back_to_filename_and_folder_when_tags_absent(tmp_path, media_cache):
    folder = tmp_path / "Folder Artist" / "Folder Album"
    folder.mkdir(parents=True)
    clip = folder / "05 - Filename Title.mp3"
    _make_clip(clip)   # no metadata tags at all
    entry = IndexEntry(id="fileid2", name=clip.name,
                        path=str(clip.relative_to(tmp_path)),
                        size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, clip, on_done, tmp_path)
    _, fields = await asyncio.wait_for(done, timeout=30)

    assert fields["display_title"] == "Filename Title"
    assert fields["track_no"] == 5
    assert fields["artist"] == "Folder Artist"
    assert fields["album"] == "Folder Album"


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_falls_back_to_artist_only_for_a_flat_top_level_dir(tmp_path, media_cache):
    """The real-world regression case, end to end through the whole pool."""
    folder = tmp_path / "Some Flat Artist"
    folder.mkdir(parents=True)
    clip = folder / "A Track.mp3"
    _make_clip(clip)   # no metadata tags at all
    entry = IndexEntry(id="fileid3", name=clip.name,
                        path=str(clip.relative_to(tmp_path)),
                        size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, clip, on_done, tmp_path)
    _, fields = await asyncio.wait_for(done, timeout=30)

    assert fields["artist"] == "Some Flat Artist"
    assert fields["album"] is None
    assert fields["artist"] != tmp_path.name, \
        "must never fall back to the shared root's own directory name"


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_uses_a_sibling_cover_file_when_no_embedded_art(tmp_path, media_cache):
    folder = tmp_path / "Some Artist" / "Some Album"
    folder.mkdir(parents=True)
    clip = folder / "01 - A Track.mp3"
    _make_clip(clip)
    (folder / "Folder.jpg").write_bytes(b"\xff\xd8\xff\xe0fake-jpeg-bytes")
    entry = IndexEntry(id="fileid4", name=clip.name,
                        path=str(clip.relative_to(tmp_path)),
                        size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, clip, on_done, tmp_path)
    _, fields = await asyncio.wait_for(done, timeout=30)

    assert fields.get("thumb_hash"), "a Folder.jpg beside the track must be picked up as its cover"
    stored = await media_cache.get_thumb(fields["thumb_hash"])
    assert stored == b"\xff\xd8\xff\xe0fake-jpeg-bytes"


async def _run(enricher, entry, path, root_path=None):
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, path, on_done, root_path)
    return await asyncio.wait_for(done, timeout=30)


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_reuses_a_cached_cover_without_rereading_it(tmp_path, media_cache):
    """
    The gap this closes: a cover already cached under this exact content
    hash used to be re-extracted (or, for a sibling file, re-read from
    disk) on every run regardless — media_cache.db had the answer and
    nothing checked it first. Proven strongly: the sibling cover file is
    deleted between the two runs, so a real second lookup would find
    nothing rather than merely redo cheap work.
    """
    folder = tmp_path / "Some Artist" / "Some Album"
    folder.mkdir(parents=True)
    clip = folder / "01 - A Track.mp3"
    _make_clip(clip)
    (folder / "Folder.jpg").write_bytes(b"\xff\xd8\xff\xe0fake-jpeg-bytes")
    entry = IndexEntry(id="fileid_reuse", name=clip.name,
                       path=str(clip.relative_to(tmp_path)),
                       size=clip.stat().st_size, type="audio", added_at=0)

    first_enricher = AudioEnricher(media_cache)
    _, first_fields = await _run(first_enricher, entry, clip, tmp_path)
    assert first_fields.get("thumb_hash")

    (folder / "Folder.jpg").unlink()
    second_enricher = AudioEnricher(media_cache)
    _, second_fields = await _run(second_enricher, entry, clip, tmp_path)

    assert second_fields["thumb_hash"] == first_fields["thumb_hash"]


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_a_moved_track_re_resolves_artist_album_with_a_cached_cover(tmp_path, media_cache):
    """
    A cached cover must never leak into stale artist/album folder context.
    Those come from _artist_album_from_ancestors against the file's
    *location*, which is exactly what a move needs re-derived
    (test_rename_reenrichment.py's audio equivalent covers the scheduling
    half) — skipping the cover lookup must not accidentally skip that too.
    """
    old_folder = tmp_path / "Old Artist" / "Old Album"
    old_folder.mkdir(parents=True)
    clip = old_folder / "01 - A Track.mp3"
    _make_clip(clip)
    (old_folder / "Folder.jpg").write_bytes(b"\xff\xd8\xff\xe0fake-jpeg-bytes")
    entry = IndexEntry(id="fileid_move", name=clip.name,
                       path=str(clip.relative_to(tmp_path)),
                       size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    _, first_fields = await _run(enricher, entry, clip, tmp_path)
    assert first_fields["artist"] == "Old Artist"
    assert first_fields.get("thumb_hash")

    new_folder = tmp_path / "New Artist" / "New Album"
    new_folder.mkdir(parents=True)
    new_clip = new_folder / clip.name
    clip.rename(new_clip)  # no cover moved along with it
    moved_entry = IndexEntry(id="fileid_move", name=new_clip.name,
                             path=str(new_clip.relative_to(tmp_path)),
                             size=entry.size, type="audio", added_at=0)
    _, second_fields = await _run(enricher, moved_entry, new_clip, tmp_path)

    assert second_fields["artist"] == "New Artist"
    assert second_fields["album"] == "New Album"
    assert second_fields["thumb_hash"] == first_fields["thumb_hash"], (
        "the cover is content-derived, not location-derived — it must still be reused")


@pytestmark_ffmpeg
def test_extract_cover_returns_none_when_no_apic_frame(tmp_path):
    from mutagen import File as MutagenFile

    clip = tmp_path / "plain.mp3"
    _make_clip(clip)

    mf = MutagenFile(str(clip))
    assert _extract_cover(mf) is None


# ── Musepack (.mpc): APEv2 tag mapping, tested against a stub object ────────
#
# mutagen has no Musepack encoder (only a decoder/tag-reader), and no
# command-line encoder was available to generate one either — there is no
# way to produce a real, valid .mpc fixture here. `_musepack_tags_from_object`
# is deliberately split out from the file-open call so the mapping itself
# can still be exercised directly, against a minimal stand-in shaped like
# mutagen's real Musepack object (an APEv2-like `.tags.get(key)` returning a
# list of values, `.info.length`).

class _StubInfo:
    def __init__(self, length):
        self.length = length


class _StubMpcFile:
    def __init__(self, tags, length=180.0):
        self.tags = tags
        self.info = _StubInfo(length)


def test_musepack_tags_from_object_reads_apev2_keys():
    stub = _StubMpcFile({
        "Title": ["A Track"], "Artist": ["Some Artist"],
        "Album": ["Some Album"], "Track": ["4"],
    })
    tags, duration = _musepack_tags_from_object(stub)
    assert tags == {"title": "A Track", "artist": "Some Artist",
                     "album": "Some Album", "track_no": 4}
    assert duration == 180.0


def test_musepack_tags_from_object_filters_a_placeholder_artist():
    stub = _StubMpcFile({"Artist": ["Unknown Artist"]})
    tags, _ = _musepack_tags_from_object(stub)
    assert "artist" not in tags


def test_musepack_tags_from_object_handles_no_tags_at_all():
    stub = _StubMpcFile(None)
    tags, duration = _musepack_tags_from_object(stub)
    assert tags == {}
    assert duration == 180.0


# ── WMA: real fixtures via ffmpeg's wmav2 encoder + asf muxer ───────────────
#
# Unlike Musepack, ffmpeg can actually produce a real, decodable .wma file,
# so this path gets genuine end-to-end coverage instead of a stub. Key
# mapping confirmed directly against both a real library sample and one of
# these fixtures before being written: `Author` (not "artist"),
# `WM/AlbumTitle`, `WM/TrackNumber` — none of them what the generic "easy"
# interface (used for every other format here) would look for.

def _make_wma_clip(path: Path, *, title=None, artist=None, album=None, track=None) -> None:
    subprocess.run(
        ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
         "-f", "lavfi", "-i", "sine=frequency=440:duration=1",
         "-c:a", "wmav2", "-b:a", "64k",
         *(["-metadata", f"title={title}"] if title else []),
         *(["-metadata", f"artist={artist}"] if artist else []),
         *(["-metadata", f"album={album}"] if album else []),
         *(["-metadata", f"track={track}"] if track else []),
         str(path)],
        check=True, capture_output=True,
    )


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_reads_wma_tags_via_the_asf_key_mapping(tmp_path, media_cache):
    clip = tmp_path / "wrong title.wma"
    _make_wma_clip(clip, title="Real Title", artist="Real Artist", album="Real Album", track=3)
    entry = IndexEntry(id="fileid5", name=clip.name, path=clip.name,
                        size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, clip, on_done)
    _, fields = await asyncio.wait_for(done, timeout=30)

    assert fields["display_title"] == "Real Title"
    assert fields["artist"] == "Real Artist"
    assert fields["album"] == "Real Album"
    assert fields["track_no"] == 3


@pytestmark_ffmpeg
@pytest.mark.asyncio
async def test_enricher_falls_back_to_folder_for_an_untagged_wma(tmp_path, media_cache):
    folder = tmp_path / "Folder Artist" / "Folder Album"
    folder.mkdir(parents=True)
    clip = folder / "A Track.wma"
    _make_wma_clip(clip)   # no metadata tags at all
    entry = IndexEntry(id="fileid6", name=clip.name,
                        path=str(clip.relative_to(tmp_path)),
                        size=clip.stat().st_size, type="audio", added_at=0)

    enricher = AudioEnricher(media_cache)
    done = asyncio.get_event_loop().create_future()

    async def on_done(file_id, fields):
        done.set_result((file_id, fields))

    enricher.spawn(entry, clip, on_done, tmp_path)
    _, fields = await asyncio.wait_for(done, timeout=30)

    assert fields["artist"] == "Folder Artist"
    assert fields["album"] == "Folder Album"