diff options
Diffstat (limited to 'packages/meshbay-node/tests')
| -rw-r--r-- | packages/meshbay-node/tests/test_enrich.py | 184 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_media_meta_request.py | 37 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_title_parse.py | 50 |
3 files changed, 265 insertions, 6 deletions
diff --git a/packages/meshbay-node/tests/test_enrich.py b/packages/meshbay-node/tests/test_enrich.py index 2205c7e..4b957cd 100644 --- a/packages/meshbay-node/tests/test_enrich.py +++ b/packages/meshbay-node/tests/test_enrich.py @@ -8,7 +8,10 @@ from pathlib import Path import pytest from meshbay_common.protocol import IndexEntry -from meshbay_node.indexer.enrich import Enricher, _season_from_ancestors, _title_from_siblings +from meshbay_node.indexer.enrich import ( + Enricher, _season_and_show_from_ancestors, _synthetic_episode_number, + _title_from_siblings, +) from meshbay_node.media_cache import MediaCache _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") @@ -16,22 +19,51 @@ _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") # ── pure helpers, no ffmpeg needed ─────────────────────────────────────────── -def test_season_from_ancestors_finds_season_folder(tmp_path): - folder = tmp_path / "Some Show" / "Season 2" +def test_season_and_show_from_ancestors_finds_season_folder(tmp_path): + show = tmp_path / "Some Show" + folder = show / "Season 2" folder.mkdir(parents=True) ep = folder / "01 - Episode Title.mkv" ep.touch() - assert _season_from_ancestors(ep) == 2 + assert _season_and_show_from_ancestors(ep) == (2, show) -def test_season_from_ancestors_none_when_no_season_folder(tmp_path): +def test_season_and_show_from_ancestors_none_when_no_season_folder(tmp_path): folder = tmp_path / "Movies" folder.mkdir() f = folder / "Some Movie 2015.mkv" f.touch() - assert _season_from_ancestors(f) is None + assert _season_and_show_from_ancestors(f) is None + + +def test_season_and_show_from_ancestors_walks_past_a_per_season_bonus_folder(tmp_path): + # The real shape this guards: a Specials/Bonus folder nested *inside* + # each numbered season folder (Show/Season N/Bonus/file.ext) rather + # than once at the show's top level — both "Bonus" and "Season N" are + # season-like on their own, and stopping at the first (innermost) + # would hand back "Season N" as the show's name instead of "Show". + show = tmp_path / "Some Show" + folder = show / "Season 1" / "Bonus" + folder.mkdir(parents=True) + f = folder / "One-Off Bonus Title.mkv" + f.touch() + + assert _season_and_show_from_ancestors(f) == (0, show) + + +def test_season_and_show_from_ancestors_book_word_season_folder(tmp_path): + # Same shape as the real bug this whole fix guards, with the "book" + # season vocabulary (§3.4) instead of "season"/"saison": a plain + # numbered-episode folder nested under it. + show = tmp_path / "Some Show" + folder = show / "Livre I" / "Episodes" + folder.mkdir(parents=True) + f = folder / "001 Episode's Own Title.mkv" + f.touch() + + assert _season_and_show_from_ancestors(f) == (1, show) def test_title_from_siblings_borrows_from_a_titled_sibling(tmp_path): @@ -53,6 +85,50 @@ def test_title_from_siblings_none_when_no_titled_sibling(tmp_path): assert _title_from_siblings(folder / "S01E02.mkv") is None +def test_title_from_siblings_ignores_a_titled_sibling_with_no_episode_number(tmp_path): + # A Specials-shaped folder: every file parses as a "movie" (its own + # one-off title, guessit finds no season/episode grammar at all) — none + # of them is a representative sibling for the show's actual name. + folder = tmp_path / "Some Show" / "Specials" + folder.mkdir(parents=True) + (folder / "Bonus Feature One.mkv").touch() + + assert _title_from_siblings(folder / "Bonus Feature Two.mkv") is None + + +def test_synthetic_episode_number_is_stable_alphabetical_rank(tmp_path): + show = tmp_path / "Some Show" + folder = show / "Specials" + folder.mkdir(parents=True) + (folder / "Bonus Feature One.mkv").touch() + (folder / "Bonus Feature Three.mkv").touch() + (folder / "Bonus Feature Two.mkv").touch() + + assert _synthetic_episode_number(folder / "Bonus Feature One.mkv", show, 0) == 1 + assert _synthetic_episode_number(folder / "Bonus Feature Three.mkv", show, 0) == 2 + assert _synthetic_episode_number(folder / "Bonus Feature Two.mkv", show, 0) == 3 + + +def test_synthetic_episode_number_does_not_collide_across_per_season_bonus_folders(tmp_path): + # The real bug this guards: season 0 spanning more than one folder + # (a Bonus folder nested inside each numbered season) — ranking within + # just one file's own folder hands out "episode 1" again in every + # other one, found live as three unrelated Specials all showing up as + # "S0E01". Ranked across the whole show instead, so each gets its own + # number. + show = tmp_path / "Some Show" + bonus1 = show / "Season 1" / "Bonus" + bonus1.mkdir(parents=True) + (bonus1 / "First Season's Bonus.mkv").touch() + bonus2 = show / "Season 2" / "Bonus" + bonus2.mkdir(parents=True) + (bonus2 / "Second Season's Bonus.mkv").touch() + + n1 = _synthetic_episode_number(bonus1 / "First Season's Bonus.mkv", show, 0) + n2 = _synthetic_episode_number(bonus2 / "Second Season's Bonus.mkv", show, 0) + assert n1 != n2 + + # ── end-to-end against a real (tiny, synthetic) video file ────────────────── pytestmark_ffmpeg = pytest.mark.skipif(not _HAVE_FFMPEG, reason="ffmpeg/ffprobe not installed") @@ -201,3 +277,99 @@ async def test_enricher_handles_episode_with_season_from_folder(tmp_path, media_ assert fields["display_title"] == "Some Show" assert fields["season"] == 3 assert fields["episode"] == 7 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_handles_specials_folder_with_no_episode_grammar(tmp_path, media_cache): + # The real bug this guards: every file in a Specials folder is named + # after its own one-off joke, not the show — guessit finds no + # season/episode in any of them, so without the ancestor-folder check + # each one used to fall to the movie branch and get searched against + # TMDB as an unrelated standalone film (found live: a real show's + # Specials folder matched several of its bonus episodes to real, + # unrelated movies that happened to share their one-off titles). + show = tmp_path / "Some Show" + season1 = show / "Season 1" + season1.mkdir(parents=True) + (season1 / "Some Show.S01E01.mkv").touch() + specials = show / "Specials" + specials.mkdir() + (specials / "Bonus Feature One.mkv").touch() + clip = specials / "Bonus Feature Two.mkv" + _make_clip(clip) + entry = IndexEntry(id="fileid3", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + # Not "Bonus Feature Two" — that's this one Special's own title, and + # matching it against TMDB by itself is exactly the bug. The show's + # name, from its ordinary season folder next door. + assert fields["display_title"] == "Some Show" + assert fields["season"] == 0 + assert fields["episode"] is not None + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_groups_bare_numbered_episodes_under_the_show_folder(tmp_path, media_cache): + """ + The real-world shape this whole fix is for: every episode (and every + Bonus feature) named "<number> <its own one-off title>", with the + show's name appearing nowhere in any filename at all — only in the + show's own root folder. guessit still finds an episode number here + (unlike the plain-Specials case above), and invents a "title" from + whatever text follows it — a different one for every file. Trusting + that per-file title, as the code used to, groups nothing together at + all: every episode becomes its own single-episode "show", searched + against TMDB by its own one-off title, and mismatched to whichever + unrelated real film or show happens to share it. + """ + show = tmp_path / "Some Show" + season1_eps = show / "Livre I" / "Episodes" + season1_eps.mkdir(parents=True) + (season1_eps / "001 First Episode's Own Title.mkv").touch() + clip = season1_eps / "002 Second Episode's Own Title.mkv" + _make_clip(clip) + season2_eps = show / "Livre II" / "Episodes" + season2_eps.mkdir(parents=True) + (season2_eps / "001 A Season 2 Episode's Own Title.mkv").touch() + entry = IndexEntry(id="fileid4", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + # Not "Second Episode's Own Title" — every episode has a different + # one, and none of them is the show. + assert fields["display_title"] == "Some Show" + assert fields["season"] == 1 + assert fields["episode"] == 2 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_reads_a_three_digit_episode_number_correctly(tmp_path, media_cache): + """ + guessit's own episode number is not trustworthy here: given a bare + leading "100", it reads that as a concatenated season+episode guess + (season=1, episode=0) rather than episode 100 — confirmed live, and + indistinguishable from a real 2-digit episode in its output. The + season-like ancestor already overrides guessit's season; this is the + same fix applied to episode. + """ + show = tmp_path / "Some Show" + season_eps = show / "Season 6" / "Episodes" + season_eps.mkdir(parents=True) + clip = season_eps / "100 A Long Season's Own Title.mkv" + _make_clip(clip) + entry = IndexEntry(id="fileid5", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + assert fields["season"] == 6 + assert fields["episode"] == 100 diff --git a/packages/meshbay-node/tests/test_media_meta_request.py b/packages/meshbay-node/tests/test_media_meta_request.py index a7355e8..a752825 100644 --- a/packages/meshbay-node/tests/test_media_meta_request.py +++ b/packages/meshbay-node/tests/test_media_meta_request.py @@ -112,6 +112,43 @@ async def test_two_episodes_in_the_same_season_folder_each_get_their_own_metadat "two different shows sharing a season folder must not resolve to the same match") +async def test_a_reclassified_file_ignores_its_stale_cached_match(media_cache): + """ + A file's classification (movie vs show) comes from its IndexEntry's + season/episode — set by index-time enrichment, which can change its + mind on a later scan (a filename-parsing fix reclassifying a whole + folder from "movie" to "tv", say) without media_cache's file->tmdb + mapping knowing anything happened: that cache is keyed by the file's + content hash alone, unchanged by any such reclassification. Found + live: exactly this scenario left every affected file answering with + its stale, wrong-kind-of-match forever, since the cache was trusted + before ever comparing media_type against what the entry resolves to + now. + """ + index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) + entry = IndexEntry(id="id-1", name="ep.mkv", path="shows/Show/Specials", + size=1, type="video", added_at=0, + display_title="Some Show", season=0, episode=1) + index.add_entry(entry) + # Pre-populate the cache exactly as it would be left over from before + # entry.season/episode existed — a movie search matched to some + # unrelated title, cached by this file's content hash. + await media_cache.set_file_tmdb("id-1", "stale-movie-id", "movie") + await media_cache.set_tmdb_meta("stale-movie-id", "movie", { + "title": "An Unrelated Movie", "original_title": "An Unrelated Movie", + "release_date": "1999-01-01", "confidence": 1.0, + }) + client = FakeTmdbClient() + session = _session(index, media_cache, client) + + await session._do_media_meta_request({"file_id": "id-1"}) + + resp = session.sent[0] + assert resp["title"] == "Some Show" + assert ("tv", "Some Show") in client.searched + assert resp["tmdb_id"] != "stale-movie-id" + + async def test_missing_file_id_is_refused(media_cache): index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) session = _session(index, media_cache, FakeTmdbClient()) diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index 1a277f2..5f0977c 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -6,6 +6,7 @@ is a manual acceptance step (§11), not something this repo's corpus holds. from meshbay_node.indexer.title_parse import ( ParsedName, + leading_episode_number, naive_title, parse_episode_filename, parse_movie_filename, @@ -111,6 +112,31 @@ def test_season_folder_roman_numeral(): assert season_from_folder_name("Saison IV") == 4 +def test_season_folder_book_word_roman_numeral(): + # Some shows number their seasons by "book" (Roman numerals) rather + # than "season"/"saison" — real, observed vocabulary (§3.4). + assert season_from_folder_name("Livre I") == 1 + assert season_from_folder_name("Livre VI") == 6 + + +def test_season_folder_bare_s_abbreviation(): + # A common abbreviated convention distinct from SEASON_WORDS' full + # words — found live: a show with "S1"/"S2"/"S3" folders instead of + # "Season 1" etc, exactly the shape that grouped its later seasons as + # loose individual entries rather than under the show. + assert season_from_folder_name("S1") == 1 + assert season_from_folder_name("S2") == 2 + assert season_from_folder_name("S02") == 2 + + +def test_season_folder_bare_s_abbreviation_does_not_match_inside_a_longer_name(): + # Anchored to the whole folder name — a real folder that merely starts + # with "S" followed by digits somewhere in a longer, unrelated name + # must not be read as a season abbreviation. + assert season_from_folder_name("S1 Extended Cut") is None + assert season_from_folder_name("Something2") is None + + def test_specials_folder_maps_to_season_zero(): assert season_from_folder_name("Specials") == 0 assert season_from_folder_name("Bonus") == 0 @@ -121,6 +147,30 @@ def test_non_season_folder_name_returns_none(): assert season_from_folder_name("Some Show Name") is None +# ── §3.4c: a bare leading episode number, guessit's 3-digit blind spot ─────── + +def test_leading_episode_number_reads_the_whole_number(): + assert leading_episode_number("001 Episode's Own Title.mkv") == 1 + assert leading_episode_number("099 Episode's Own Title.mkv") == 99 + + +def test_leading_episode_number_not_split_by_guessit_at_three_digits(): + # The real bug this guards: guessit itself reads a bare "100" as season=1, + # episode=0 (a concatenated SxxE guess), not episode=100 — silently, with + # nothing in its output telling apart a real 2-digit episode from this. + assert leading_episode_number("100 Episode's Own Title.mkv") == 100 + + +def test_leading_episode_number_ignores_a_leading_year(): + # Four digits, not the 1-3 an episode-number prefix can be — this is a + # normal movie filename shape, not an episode number. + assert leading_episode_number("2010 - Some Movie.mkv") is None + + +def test_leading_episode_number_none_without_a_leading_number(): + assert leading_episode_number("Some Show.S01E01.mkv") is None + + def test_parsed_name_is_a_plain_dataclass(): # sanity: constructible with just the one required field, per the # "None => caller must supply from elsewhere" contract. |