diff options
| author | Christophe Besson <cbesson@gmail.com> | 2026-08-26 17:18:37 +0200 |
|---|---|---|
| committer | Christophe Besson <cbesson@gmail.com> | 2026-08-26 17:18:37 +0200 |
| commit | 80cb6c4dc5f22336387f2ac74ef2cb85cf2cf7d4 (patch) | |
| tree | 590e2daf81b2feef674f1bd580b9bf40f83addad /packages/meshbay-node | |
| parent | ca62b4123b854fc9ece258e14c84c895aaa275cd (diff) | |
| parent | 59289b82dc8af08c605fd247c90a5c5ec92c6db3 (diff) | |
| download | meshbay-80cb6c4dc5f22336387f2ac74ef2cb85cf2cf7d4.tar.gz | |
Merge branch 'feat/video-type-filter'
Videos app grouping and TMDB matching fixes, plus a new toolbar filter:
- A season-like ancestor folder (numbered season, or Specials/Bonus/Extras
-> season 0) now names the show from its own root folder, unconditionally
— never a per-file guessit title, which cannot tell a show's real name
from an individual episode's own one-off name when the filename carries
no reliable ShowName/SxxExx structure. Fixes a real show's episodes and
Specials folder alike showing up as dozens of individual "movies", each
matched against TMDB by its own one-off title.
- The ancestor walk continues past every consecutive season-like folder,
not just the first — a per-season Bonus folder is nested two levels
inside the show, and stopping at the first would name the season as the
show.
- A season spanning more than one folder (a per-book Bonus folder nested
inside every numbered season) no longer hands out colliding episode
numbers independently in each one.
- guessit's own episode number is not trusted when it comes from a bare
3-digit leading number ("100" parses as season=1/episode=0, not
episode=100) — read directly via regex instead.
- Recognizes "S1"/"S2"-style abbreviated season folders, not just full
words ("Season"/"Saison"/"Livre") — a real show organized its later
seasons this way and they never got the ancestor-based grouping fix at
all.
- The TMDB match cache is no longer trusted across a movie<->show
reclassification it doesn't know happened.
- New: an All/Movies/Series filter in the Videos toolbar, to the left of
the search field, defaulting to "All"; wraps correctly on mobile.
Verified live against the real libraries these were found on throughout —
not synthetic reproduction alone. 430 hub tests + 667 node tests passing.
Diffstat (limited to 'packages/meshbay-node')
6 files changed, 439 insertions, 23 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index ae3f3dc..07f0be7 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -36,17 +36,52 @@ MAX_ANCESTOR_DEPTH = 4 # Bounds the "borrow a title from a sibling episode filename" scan (§3.4) so # a folder with thousands of files costs a fixed, small amount of work. MAX_SIBLINGS_CHECKED = 20 +# Bounds the whole-show scan _synthetic_episode_number uses to rank a +# season's files across more than one folder (§3.4c) — a show's total file +# count, not just one folder's, so this needs more headroom than +# MAX_SIBLINGS_CHECKED. +MAX_SEASON_FILES_CHECKED = 500 -def _season_from_ancestors(file_path: Path) -> int | None: +def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, Path] | None: + """ + §3.4/§3.4c: walk up ancestor folders for a season-like one (a numbered + season, or Specials/Bonus/Extras -> season 0) — season from the first + (innermost) match, but the show's own name from *above every + consecutive season-like ancestor*, not just the first one. A + per-season Bonus folder (`Show/Season N/Bonus/file.ext`) is nested two + levels inside the show, both of them season-like on their own + ("Bonus" and "Season N") — stopping at the first would hand back + "Season N" as the show's name instead of "Show". + + Trusted over any per-file guessit title once found: a bare episode + numbering convention with no show name embedded at all + (`001 Episode's Own Title.mkv`, no SxxExx, no show prefix) is + completely ordinary and makes guessit invent a title from whatever + text follows the number — that text is the individual episode's own + name, never the show's, and every episode in the folder produces a + different one. None of that ambiguity exists for the folder structure + itself: the show's own root folder names it once, not per file. + + None if no ancestor looks like a season folder at all — a flat + library has nothing to borrow a show name from this way, and + _title_from_siblings' per-file logic is what applies instead. + """ folder = file_path.parent - for _ in range(MAX_ANCESTOR_DEPTH): - if folder is None or folder == folder.parent: - break - season = title_parse.season_from_folder_name(folder.name) - if season is not None: - return season + season: int | None = None + depth = 0 + while folder is not None and folder.parent != folder and depth < MAX_ANCESTOR_DEPTH: + this_season = title_parse.season_from_folder_name(folder.name) + if this_season is None: + if season is None: + folder = folder.parent + depth += 1 + continue + return season, folder + if season is None: + season = this_season folder = folder.parent + depth += 1 return None @@ -55,6 +90,13 @@ def _title_from_siblings(file_path: Path) -> str | None: §3.4: an episode filename with no show name in it borrows the title from a representative sibling in the same folder, never from the folder name alone (an acronym-named show folder is a real, observed case). + + Requires the sibling to carry its own episode number too, not just a + title — a folder where every file is a one-off-named Special (§3.4b) + has plenty of `display_title`s (guessit reads *a* title off nearly + anything) but none of them name the show; requiring a real episode + number alongside is what tells apart a genuinely representative sibling + from another Special just like this one. """ try: names = sorted(p.name for p in file_path.parent.iterdir() if p.is_file()) @@ -70,11 +112,52 @@ def _title_from_siblings(file_path: Path) -> str | None: if checked > MAX_SIBLINGS_CHECKED: break parsed = title_parse.parse_episode_filename(name) - if parsed.display_title: + if parsed.display_title and parsed.episode is not None: return parsed.display_title return None +def _synthetic_episode_number(file_path: Path, show_root: Path, season: int) -> int: + """ + §3.4b/§3.4c: a Specials/Bonus folder's files often carry no episode + number at all — each is just named after its own one-off title. The + frontend (video-app.js's buildSeasons) sorts within a season by this + number but only needs it to provide a stable order, not to mean + anything beyond that. + + Ranked across the *whole show*, not just this file's own folder: season + 0 routinely spans more than one folder under the show's root — a + Bonus folder nested inside every numbered season + (`Show/Season N/Bonus/...`) is exactly the shape that produces this — + and ranking within just this file's own folder hands out the same + "episode 1" again in every other one, found live as three unrelated + Specials all showing up as "S0E01". Deterministic across re-scans as + long as the show's contents don't change. 1-based so it reads as + "episode 1", not "episode 0", in a UI that already uses season 0 for + "Specials" itself. Bounded (MAX_SEASON_FILES_CHECKED) so a huge show + costs a fixed amount of work rather than scaling with its size. + """ + try: + candidates = sorted( + (p for p in show_root.rglob("*") + if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]), + key=lambda p: (str(p.parent), p.name), + ) + except OSError: + return 1 + rank = 0 + for i, candidate in enumerate(candidates): + if i >= MAX_SEASON_FILES_CHECKED: + break + ancestor = _season_and_show_from_ancestors(candidate) + if ancestor is None or ancestor[0] != season: + continue + rank += 1 + if candidate == file_path: + return rank + return 1 + + async def _make_thumbnail(file_path: Path, duration: float | None) -> bytes | None: """One ffmpeg frame grab at ~10% of duration (or 5s if unknown), scaled down.""" seek = max(0.0, (duration or 50.0) * 0.1) @@ -165,14 +248,48 @@ class Enricher: entry.id, fields.get("duration"), fields.get("width"), fields.get("height")) ep = title_parse.parse_episode_filename(entry.name) - if ep.episode is not None: + ancestor = await asyncio.to_thread(_season_and_show_from_ancestors, file_path) + + if ancestor is not None: + # A season-like ancestor folder exists — the show's own + # root folder names it, trusted over any per-file guessit + # title. This is what actually groups every episode under + # one show: a bare numbering convention with no show name + # in the filename at all (`001 Episode's Own Title.mkv`, + # ordinary enough on its own) makes guessit invent a title + # from whatever text follows the number, which is that + # episode's own name, never the show's — and differs for + # every episode, so nothing would ever group together. + # Found live: a real show's episodes and its Specials + # folder alike, both named this way, showed up as + # individual "movies" each searched against — and matched + # to — an unrelated real film sharing that one-off title, + # instead of anything grouped under the show at all. + season, show_folder = ancestor + # Not ep.episode: guessit reads a bare 3-digit leading + # number (routine once a season-like ancestor is already + # doing the real season/episode work) as a concatenated + # season+episode guess rather than a plain episode number + # — confirmed live, "100 Title.mkv" parsed as episode 0, + # not 100, with nothing in guessit's own output telling + # that apart from a real 2-digit episode. The direct regex + # reads the whole leading number as it is. + episode = title_parse.leading_episode_number(entry.name) + if episode is None: + episode = ep.episode + if episode is None: + episode = await asyncio.to_thread( + _synthetic_episode_number, file_path, show_folder, season) + fields["display_title"] = show_folder.name + fields["season"] = season + fields["episode"] = episode + elif ep.episode is not None: + # No season-like ancestor at all (a flat library) but the + # filename itself carries season+episode (§3.4). title = ep.display_title or await asyncio.to_thread( _title_from_siblings, file_path) - season = ep.season - if season is None: - season = await asyncio.to_thread(_season_from_ancestors, file_path) fields["display_title"] = title or title_parse.naive_title(entry.name) - fields["season"] = season + fields["season"] = ep.season fields["episode"] = ep.episode else: mv = title_parse.parse_movie_filename(entry.name) diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 5203247..24b423e 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -36,13 +36,21 @@ _EDITION_RE = re.compile("|".join(_EDITION_PHRASES), re.IGNORECASE) # A season-like ancestor folder: the English/French words plus a number or # Roman numeral. Vocabulary is a plain tuple so a deployment can extend it -# per locale without touching the regex-building logic. -SEASON_WORDS = ("season", "saison") +# per locale without touching the regex-building logic. "livre" ("book") is +# real, observed vocabulary too — some shows name their seasons that way +# (Roman numerals: "Livre I".."Livre VI") rather than "saison". +SEASON_WORDS = ("season", "saison", "livre") _SEASON_RE = re.compile( r"(?:" + "|".join(SEASON_WORDS) + r")\s*([0-9]+|[ivxlc]+)\b", re.IGNORECASE, ) _SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE) +# A bare "S" + number as the *whole* folder name — "S1", "S2", "S02" — a +# common abbreviated convention distinct from SEASON_WORDS' full words. +# Anchored to the entire name, not just `\b`-bounded within a longer +# string, so it only matches a folder actually named just that — never +# some other word that merely starts with "s" followed by digits. +_SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE) _ROMAN_NUMERALS = { 2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI", @@ -108,6 +116,9 @@ def season_from_folder_name(name: str) -> int | None: """ if _SPECIALS_RE.search(name): return 0 + m = _SEASON_ABBREV_RE.match(name.strip()) + if m: + return int(m.group(1)) m = _SEASON_RE.search(name) if not m: return None @@ -227,3 +238,20 @@ def parse_episode_filename(filename: str) -> ParsedName: display_title=title or None, naive_title=nt, season=season, episode=episode, confidence=confidence, ) + + +# A bare leading episode number, no show name attached (§3.4c) — the same +# shape as music's _TRACK_PREFIX_RE, capped at 3 digits for the same reason: +# a leading year ("2010 - Episode.mkv") is 4 digits and must not match. +# guessit's own `episode` is not a substitute here: given exactly 3 digits it +# tries to read them as a concatenated SxxE/SEE season+episode pair instead +# of a plain episode number — confirmed live, "100 Title.mkv" parses as +# season=1, episode=0, not episode=100 — silently wrong in a way nothing +# about its output distinguishes from a real 2-digit episode. This reads the +# whole leading number as one value instead. +_LEADING_NUMBER_RE = re.compile(r"^(\d{1,3})[\s._-]+(?=\S)") + + +def leading_episode_number(filename: str) -> int | None: + m = _LEADING_NUMBER_RE.match(filename) + return int(m.group(1)) if m else None diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py index f95e59e..252e202 100644 --- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py +++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py @@ -2914,8 +2914,20 @@ class WebRTCPeerSession: meta = None tmdb_id = None if cached is not None: - tmdb_id, media_type = cached - meta = await media_cache.get_tmdb_meta(tmdb_id, media_type) + cached_tmdb_id, cached_media_type = cached + # Trustworthy only if it still agrees with what this file + # resolves to *now*. season/episode come from index-time + # enrichment (enrich.py), which can reclassify a file between + # movie and show on a later scan without this cache knowing — + # it is keyed by the file's content hash alone, which a + # reclassification never changes. Found live: an enrichment fix + # to a Specials-folder bug reclassified hundreds of files from + # "movie" to "tv", and every one kept answering with its + # stale movie-era match forever, because this was trusted + # before ever comparing media_type against the current one. + if cached_media_type == media_type: + tmdb_id = cached_tmdb_id + meta = await media_cache.get_tmdb_meta(tmdb_id, media_type) if meta is None: result, ratio = await self._tmdb_search(tmdb_client, entry, is_show) diff --git a/packages/meshbay-node/tests/test_enrich.py b/packages/meshbay-node/tests/test_enrich.py index 2205c7e..4b957cd 100644 --- a/packages/meshbay-node/tests/test_enrich.py +++ b/packages/meshbay-node/tests/test_enrich.py @@ -8,7 +8,10 @@ from pathlib import Path import pytest from meshbay_common.protocol import IndexEntry -from meshbay_node.indexer.enrich import Enricher, _season_from_ancestors, _title_from_siblings +from meshbay_node.indexer.enrich import ( + Enricher, _season_and_show_from_ancestors, _synthetic_episode_number, + _title_from_siblings, +) from meshbay_node.media_cache import MediaCache _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") @@ -16,22 +19,51 @@ _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") # ── pure helpers, no ffmpeg needed ─────────────────────────────────────────── -def test_season_from_ancestors_finds_season_folder(tmp_path): - folder = tmp_path / "Some Show" / "Season 2" +def test_season_and_show_from_ancestors_finds_season_folder(tmp_path): + show = tmp_path / "Some Show" + folder = show / "Season 2" folder.mkdir(parents=True) ep = folder / "01 - Episode Title.mkv" ep.touch() - assert _season_from_ancestors(ep) == 2 + assert _season_and_show_from_ancestors(ep) == (2, show) -def test_season_from_ancestors_none_when_no_season_folder(tmp_path): +def test_season_and_show_from_ancestors_none_when_no_season_folder(tmp_path): folder = tmp_path / "Movies" folder.mkdir() f = folder / "Some Movie 2015.mkv" f.touch() - assert _season_from_ancestors(f) is None + assert _season_and_show_from_ancestors(f) is None + + +def test_season_and_show_from_ancestors_walks_past_a_per_season_bonus_folder(tmp_path): + # The real shape this guards: a Specials/Bonus folder nested *inside* + # each numbered season folder (Show/Season N/Bonus/file.ext) rather + # than once at the show's top level — both "Bonus" and "Season N" are + # season-like on their own, and stopping at the first (innermost) + # would hand back "Season N" as the show's name instead of "Show". + show = tmp_path / "Some Show" + folder = show / "Season 1" / "Bonus" + folder.mkdir(parents=True) + f = folder / "One-Off Bonus Title.mkv" + f.touch() + + assert _season_and_show_from_ancestors(f) == (0, show) + + +def test_season_and_show_from_ancestors_book_word_season_folder(tmp_path): + # Same shape as the real bug this whole fix guards, with the "book" + # season vocabulary (§3.4) instead of "season"/"saison": a plain + # numbered-episode folder nested under it. + show = tmp_path / "Some Show" + folder = show / "Livre I" / "Episodes" + folder.mkdir(parents=True) + f = folder / "001 Episode's Own Title.mkv" + f.touch() + + assert _season_and_show_from_ancestors(f) == (1, show) def test_title_from_siblings_borrows_from_a_titled_sibling(tmp_path): @@ -53,6 +85,50 @@ def test_title_from_siblings_none_when_no_titled_sibling(tmp_path): assert _title_from_siblings(folder / "S01E02.mkv") is None +def test_title_from_siblings_ignores_a_titled_sibling_with_no_episode_number(tmp_path): + # A Specials-shaped folder: every file parses as a "movie" (its own + # one-off title, guessit finds no season/episode grammar at all) — none + # of them is a representative sibling for the show's actual name. + folder = tmp_path / "Some Show" / "Specials" + folder.mkdir(parents=True) + (folder / "Bonus Feature One.mkv").touch() + + assert _title_from_siblings(folder / "Bonus Feature Two.mkv") is None + + +def test_synthetic_episode_number_is_stable_alphabetical_rank(tmp_path): + show = tmp_path / "Some Show" + folder = show / "Specials" + folder.mkdir(parents=True) + (folder / "Bonus Feature One.mkv").touch() + (folder / "Bonus Feature Three.mkv").touch() + (folder / "Bonus Feature Two.mkv").touch() + + assert _synthetic_episode_number(folder / "Bonus Feature One.mkv", show, 0) == 1 + assert _synthetic_episode_number(folder / "Bonus Feature Three.mkv", show, 0) == 2 + assert _synthetic_episode_number(folder / "Bonus Feature Two.mkv", show, 0) == 3 + + +def test_synthetic_episode_number_does_not_collide_across_per_season_bonus_folders(tmp_path): + # The real bug this guards: season 0 spanning more than one folder + # (a Bonus folder nested inside each numbered season) — ranking within + # just one file's own folder hands out "episode 1" again in every + # other one, found live as three unrelated Specials all showing up as + # "S0E01". Ranked across the whole show instead, so each gets its own + # number. + show = tmp_path / "Some Show" + bonus1 = show / "Season 1" / "Bonus" + bonus1.mkdir(parents=True) + (bonus1 / "First Season's Bonus.mkv").touch() + bonus2 = show / "Season 2" / "Bonus" + bonus2.mkdir(parents=True) + (bonus2 / "Second Season's Bonus.mkv").touch() + + n1 = _synthetic_episode_number(bonus1 / "First Season's Bonus.mkv", show, 0) + n2 = _synthetic_episode_number(bonus2 / "Second Season's Bonus.mkv", show, 0) + assert n1 != n2 + + # ── end-to-end against a real (tiny, synthetic) video file ────────────────── pytestmark_ffmpeg = pytest.mark.skipif(not _HAVE_FFMPEG, reason="ffmpeg/ffprobe not installed") @@ -201,3 +277,99 @@ async def test_enricher_handles_episode_with_season_from_folder(tmp_path, media_ assert fields["display_title"] == "Some Show" assert fields["season"] == 3 assert fields["episode"] == 7 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_handles_specials_folder_with_no_episode_grammar(tmp_path, media_cache): + # The real bug this guards: every file in a Specials folder is named + # after its own one-off joke, not the show — guessit finds no + # season/episode in any of them, so without the ancestor-folder check + # each one used to fall to the movie branch and get searched against + # TMDB as an unrelated standalone film (found live: a real show's + # Specials folder matched several of its bonus episodes to real, + # unrelated movies that happened to share their one-off titles). + show = tmp_path / "Some Show" + season1 = show / "Season 1" + season1.mkdir(parents=True) + (season1 / "Some Show.S01E01.mkv").touch() + specials = show / "Specials" + specials.mkdir() + (specials / "Bonus Feature One.mkv").touch() + clip = specials / "Bonus Feature Two.mkv" + _make_clip(clip) + entry = IndexEntry(id="fileid3", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + # Not "Bonus Feature Two" — that's this one Special's own title, and + # matching it against TMDB by itself is exactly the bug. The show's + # name, from its ordinary season folder next door. + assert fields["display_title"] == "Some Show" + assert fields["season"] == 0 + assert fields["episode"] is not None + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_groups_bare_numbered_episodes_under_the_show_folder(tmp_path, media_cache): + """ + The real-world shape this whole fix is for: every episode (and every + Bonus feature) named "<number> <its own one-off title>", with the + show's name appearing nowhere in any filename at all — only in the + show's own root folder. guessit still finds an episode number here + (unlike the plain-Specials case above), and invents a "title" from + whatever text follows it — a different one for every file. Trusting + that per-file title, as the code used to, groups nothing together at + all: every episode becomes its own single-episode "show", searched + against TMDB by its own one-off title, and mismatched to whichever + unrelated real film or show happens to share it. + """ + show = tmp_path / "Some Show" + season1_eps = show / "Livre I" / "Episodes" + season1_eps.mkdir(parents=True) + (season1_eps / "001 First Episode's Own Title.mkv").touch() + clip = season1_eps / "002 Second Episode's Own Title.mkv" + _make_clip(clip) + season2_eps = show / "Livre II" / "Episodes" + season2_eps.mkdir(parents=True) + (season2_eps / "001 A Season 2 Episode's Own Title.mkv").touch() + entry = IndexEntry(id="fileid4", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + # Not "Second Episode's Own Title" — every episode has a different + # one, and none of them is the show. + assert fields["display_title"] == "Some Show" + assert fields["season"] == 1 + assert fields["episode"] == 2 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_reads_a_three_digit_episode_number_correctly(tmp_path, media_cache): + """ + guessit's own episode number is not trustworthy here: given a bare + leading "100", it reads that as a concatenated season+episode guess + (season=1, episode=0) rather than episode 100 — confirmed live, and + indistinguishable from a real 2-digit episode in its output. The + season-like ancestor already overrides guessit's season; this is the + same fix applied to episode. + """ + show = tmp_path / "Some Show" + season_eps = show / "Season 6" / "Episodes" + season_eps.mkdir(parents=True) + clip = season_eps / "100 A Long Season's Own Title.mkv" + _make_clip(clip) + entry = IndexEntry(id="fileid5", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + assert fields["season"] == 6 + assert fields["episode"] == 100 diff --git a/packages/meshbay-node/tests/test_media_meta_request.py b/packages/meshbay-node/tests/test_media_meta_request.py index a7355e8..a752825 100644 --- a/packages/meshbay-node/tests/test_media_meta_request.py +++ b/packages/meshbay-node/tests/test_media_meta_request.py @@ -112,6 +112,43 @@ async def test_two_episodes_in_the_same_season_folder_each_get_their_own_metadat "two different shows sharing a season folder must not resolve to the same match") +async def test_a_reclassified_file_ignores_its_stale_cached_match(media_cache): + """ + A file's classification (movie vs show) comes from its IndexEntry's + season/episode — set by index-time enrichment, which can change its + mind on a later scan (a filename-parsing fix reclassifying a whole + folder from "movie" to "tv", say) without media_cache's file->tmdb + mapping knowing anything happened: that cache is keyed by the file's + content hash alone, unchanged by any such reclassification. Found + live: exactly this scenario left every affected file answering with + its stale, wrong-kind-of-match forever, since the cache was trusted + before ever comparing media_type against what the entry resolves to + now. + """ + index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) + entry = IndexEntry(id="id-1", name="ep.mkv", path="shows/Show/Specials", + size=1, type="video", added_at=0, + display_title="Some Show", season=0, episode=1) + index.add_entry(entry) + # Pre-populate the cache exactly as it would be left over from before + # entry.season/episode existed — a movie search matched to some + # unrelated title, cached by this file's content hash. + await media_cache.set_file_tmdb("id-1", "stale-movie-id", "movie") + await media_cache.set_tmdb_meta("stale-movie-id", "movie", { + "title": "An Unrelated Movie", "original_title": "An Unrelated Movie", + "release_date": "1999-01-01", "confidence": 1.0, + }) + client = FakeTmdbClient() + session = _session(index, media_cache, client) + + await session._do_media_meta_request({"file_id": "id-1"}) + + resp = session.sent[0] + assert resp["title"] == "Some Show" + assert ("tv", "Some Show") in client.searched + assert resp["tmdb_id"] != "stale-movie-id" + + async def test_missing_file_id_is_refused(media_cache): index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) session = _session(index, media_cache, FakeTmdbClient()) diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index 1a277f2..5f0977c 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -6,6 +6,7 @@ is a manual acceptance step (§11), not something this repo's corpus holds. from meshbay_node.indexer.title_parse import ( ParsedName, + leading_episode_number, naive_title, parse_episode_filename, parse_movie_filename, @@ -111,6 +112,31 @@ def test_season_folder_roman_numeral(): assert season_from_folder_name("Saison IV") == 4 +def test_season_folder_book_word_roman_numeral(): + # Some shows number their seasons by "book" (Roman numerals) rather + # than "season"/"saison" — real, observed vocabulary (§3.4). + assert season_from_folder_name("Livre I") == 1 + assert season_from_folder_name("Livre VI") == 6 + + +def test_season_folder_bare_s_abbreviation(): + # A common abbreviated convention distinct from SEASON_WORDS' full + # words — found live: a show with "S1"/"S2"/"S3" folders instead of + # "Season 1" etc, exactly the shape that grouped its later seasons as + # loose individual entries rather than under the show. + assert season_from_folder_name("S1") == 1 + assert season_from_folder_name("S2") == 2 + assert season_from_folder_name("S02") == 2 + + +def test_season_folder_bare_s_abbreviation_does_not_match_inside_a_longer_name(): + # Anchored to the whole folder name — a real folder that merely starts + # with "S" followed by digits somewhere in a longer, unrelated name + # must not be read as a season abbreviation. + assert season_from_folder_name("S1 Extended Cut") is None + assert season_from_folder_name("Something2") is None + + def test_specials_folder_maps_to_season_zero(): assert season_from_folder_name("Specials") == 0 assert season_from_folder_name("Bonus") == 0 @@ -121,6 +147,30 @@ def test_non_season_folder_name_returns_none(): assert season_from_folder_name("Some Show Name") is None +# ── §3.4c: a bare leading episode number, guessit's 3-digit blind spot ─────── + +def test_leading_episode_number_reads_the_whole_number(): + assert leading_episode_number("001 Episode's Own Title.mkv") == 1 + assert leading_episode_number("099 Episode's Own Title.mkv") == 99 + + +def test_leading_episode_number_not_split_by_guessit_at_three_digits(): + # The real bug this guards: guessit itself reads a bare "100" as season=1, + # episode=0 (a concatenated SxxE guess), not episode=100 — silently, with + # nothing in its output telling apart a real 2-digit episode from this. + assert leading_episode_number("100 Episode's Own Title.mkv") == 100 + + +def test_leading_episode_number_ignores_a_leading_year(): + # Four digits, not the 1-3 an episode-number prefix can be — this is a + # normal movie filename shape, not an episode number. + assert leading_episode_number("2010 - Some Movie.mkv") is None + + +def test_leading_episode_number_none_without_a_leading_number(): + assert leading_episode_number("Some Show.S01E01.mkv") is None + + def test_parsed_name_is_a_plain_dataclass(): # sanity: constructible with just the one required field, per the # "None => caller must supply from elsewhere" contract. |