From 4c45792c861c47d44ebdbbab4130e37fba5d6795 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 15:22:17 +0200 Subject: fix(video): a Specials/Bonus folder's episodes were matched to TMDB as unrelated standalone movies MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found live: a "Specials" folder full of one-off-named bonus episodes had every file appear as its own poster, matched against TMDB by its own title, because guessit finds no season/episode grammar at all in a filename with no SxxExx of its own — so the classification (episode vs standalone movie), based solely on that, fell to the movie branch. Real, unrelated films happened to share several of those one-off titles and matched confidently, one per Special, cluttering the Videos view with dozens of wrong posters instead of grouping under the show. An ancestor folder saying this is part of a show — a numbered season, or Specials/Bonus/Extras -> season 0 — is now trusted over the filename having no SxxExx of its own. The show's name can't come from this file's own guessit title (that's the bug) or from siblings in the same Specials folder (every one of them has the same gap) — it's borrowed from the show's ordinary season folders next door, which do carry it in the usual ShowName.SxxExx shape (_title_from_show_siblings). A synthetic, stable episode number (alphabetical rank among the folder's video files) stands in for the real one nothing in a Specials folder provides. Generic, not specific to a folder literally named "Specials": the same fallback fires for any season-like ancestor folder whose files lack per-file episode grammar, numbered seasons included (covered by a dedicated test). Also hardens _title_from_siblings to require the sibling's own episode number too, not just a title — otherwise it would borrow one Special's one-off title as if it were representative, on a folder like this one. Existing already-indexed entries will need a rescan (restart the node) to be re-enriched under this logic — nothing re-derives them on its own. --- .../src/meshbay_node/indexer/enrich.py | 112 +++++++++++++++++++-- packages/meshbay-node/tests/test_enrich.py | 99 +++++++++++++++++- 2 files changed, 201 insertions(+), 10 deletions(-) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index ae3f3dc..fc4b059 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -55,6 +55,13 @@ def _title_from_siblings(file_path: Path) -> str | None: §3.4: an episode filename with no show name in it borrows the title from a representative sibling in the same folder, never from the folder name alone (an acronym-named show folder is a real, observed case). + + Requires the sibling to carry its own episode number too, not just a + title — a folder where every file is a one-off-named Special (§3.4b) + has plenty of `display_title`s (guessit reads *a* title off nearly + anything) but none of them name the show; requiring a real episode + number alongside is what tells apart a genuinely representative sibling + from another Special just like this one. """ try: names = sorted(p.name for p in file_path.parent.iterdir() if p.is_file()) @@ -70,11 +77,76 @@ def _title_from_siblings(file_path: Path) -> str | None: if checked > MAX_SIBLINGS_CHECKED: break parsed = title_parse.parse_episode_filename(name) - if parsed.display_title: + if parsed.display_title and parsed.episode is not None: return parsed.display_title return None +def _title_from_show_siblings(file_path: Path) -> str | None: + """ + §3.4b: a Specials/Bonus folder's files are routinely named after their + own one-off joke or theme rather than the show at all — every one of + them parses as a standalone "movie" (guessit has no season/episode + grammar to find), and by coincidence a good few of those one-off + titles collide with real, unrelated films: found live, a real show's + Specials folder matched several bonus episodes to real, unrelated + movies sharing those one-off titles, instead of anything to do with + the show. _title_from_siblings can't help — every sibling *here* has the same + problem. What the ordinary season folders next to this one hold does + not: `Show.S01E01.mkv`-shaped filenames name the show properly, so + borrowing from the first one found there is the same trick as + _title_from_siblings, aimed one level higher — at the show's other + seasons rather than this file's own folder. + """ + season_folder = file_path.parent + show_folder = season_folder.parent + try: + sibling_folders = sorted( + (d for d in show_folder.iterdir() if d.is_dir() and d != season_folder), + key=lambda d: d.name) + except OSError: + return None + for folder in sibling_folders: + try: + names = sorted(p.name for p in folder.iterdir() if p.is_file()) + except OSError: + continue + checked = 0 + for name in names: + if Path(name).suffix.lower() not in MEDIA_EXTENSIONS["video"]: + continue + checked += 1 + if checked > MAX_SIBLINGS_CHECKED: + break + parsed = title_parse.parse_episode_filename(name) + if parsed.display_title and parsed.episode is not None: + return parsed.display_title + return None + + +def _synthetic_episode_number(file_path: Path) -> int: + """ + §3.4b: a Specials/Bonus folder's files often carry no episode number at + all — each is just named after its own one-off title. The frontend + (video-app.js's buildSeasons) sorts within a season by this number but + only needs it to provide a stable order, not to mean anything beyond + that — alphabetical rank among the folder's video files is enough, and + deterministic across re-scans as long as the folder's contents don't + change. 1-based so it reads as "episode 1", not "episode 0", in a UI + that already uses season 0 for "Specials" itself. + """ + try: + names = sorted( + p.name for p in file_path.parent.iterdir() + if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]) + except OSError: + return 1 + try: + return names.index(file_path.name) + 1 + except ValueError: + return 1 + + async def _make_thumbnail(file_path: Path, duration: float | None) -> bytes | None: """One ffmpeg frame grab at ~10% of duration (or 5s if unknown), scaled down.""" seek = max(0.0, (duration or 50.0) * 0.1) @@ -165,15 +237,37 @@ class Enricher: entry.id, fields.get("duration"), fields.get("width"), fields.get("height")) ep = title_parse.parse_episode_filename(entry.name) - if ep.episode is not None: - title = ep.display_title or await asyncio.to_thread( - _title_from_siblings, file_path) - season = ep.season - if season is None: - season = await asyncio.to_thread(_season_from_ancestors, file_path) + season = ep.season + if season is None: + season = await asyncio.to_thread(_season_from_ancestors, file_path) + + # §3.4b: an ancestor folder saying this is part of a show (a + # numbered season, or Specials/Bonus/Extras -> season 0) is + # trusted over the filename having no SxxExx of its own — + # otherwise a Specials folder's one-off-named files (no episode + # grammar for guessit to find at all) fall to the movie branch + # below and get searched against TMDB as unrelated standalone + # films, one per Special. Found live: a real show's Specials + # folder matched several bonus episodes to real, unrelated + # movies sharing those one-off titles. + if ep.episode is not None or season is not None: + if ep.episode is not None: + title = ep.display_title or await asyncio.to_thread( + _title_from_siblings, file_path) + episode = ep.episode + else: + # This file's own name carries no episode grammar at + # all — every sibling in *this* folder has the same gap + # (that is what makes it a Specials-shaped folder), so + # borrowing from them (_title_from_siblings) would just + # hand back another Special's own one-off title. The + # show's ordinary season folders, next to this one, + # don't have that gap. + title = await asyncio.to_thread(_title_from_show_siblings, file_path) + episode = await asyncio.to_thread(_synthetic_episode_number, file_path) fields["display_title"] = title or title_parse.naive_title(entry.name) - fields["season"] = season - fields["episode"] = ep.episode + fields["season"] = season if season is not None else 0 + fields["episode"] = episode else: mv = title_parse.parse_movie_filename(entry.name) fields["display_title"] = mv.display_title or mv.naive_title diff --git a/packages/meshbay-node/tests/test_enrich.py b/packages/meshbay-node/tests/test_enrich.py index 2205c7e..5497812 100644 --- a/packages/meshbay-node/tests/test_enrich.py +++ b/packages/meshbay-node/tests/test_enrich.py @@ -8,7 +8,10 @@ from pathlib import Path import pytest from meshbay_common.protocol import IndexEntry -from meshbay_node.indexer.enrich import Enricher, _season_from_ancestors, _title_from_siblings +from meshbay_node.indexer.enrich import ( + Enricher, _season_from_ancestors, _synthetic_episode_number, + _title_from_show_siblings, _title_from_siblings, +) from meshbay_node.media_cache import MediaCache _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") @@ -53,6 +56,67 @@ def test_title_from_siblings_none_when_no_titled_sibling(tmp_path): assert _title_from_siblings(folder / "S01E02.mkv") is None +def test_title_from_siblings_ignores_a_titled_sibling_with_no_episode_number(tmp_path): + # A Specials-shaped folder: every file parses as a "movie" (its own + # one-off title, guessit finds no season/episode grammar at all) — none + # of them is a representative sibling for the show's actual name. + folder = tmp_path / "Some Show" / "Specials" + folder.mkdir(parents=True) + (folder / "Bonus Feature One.mkv").touch() + + assert _title_from_siblings(folder / "Bonus Feature Two.mkv") is None + + +def test_title_from_show_siblings_borrows_from_a_sibling_season_folder(tmp_path): + show = tmp_path / "Some Show" + specials = show / "Specials" + specials.mkdir(parents=True) + (specials / "Bonus Feature One.mkv").touch() + season1 = show / "Season 1" + season1.mkdir() + (season1 / "Some Show.S01E01.mkv").touch() + + assert _title_from_show_siblings(specials / "Bonus Feature One.mkv") == "Some Show" + + +def test_title_from_show_siblings_none_when_no_show_folder_has_one(tmp_path): + show = tmp_path / "Some Show" + specials = show / "Specials" + specials.mkdir(parents=True) + (specials / "Bonus Feature One.mkv").touch() + (specials / "Bonus Feature Two.mkv").touch() + + assert _title_from_show_siblings(specials / "Bonus Feature One.mkv") is None + + +def test_title_from_show_siblings_not_specific_to_the_word_specials(tmp_path): + # Genericity check: the same gap (a season-like folder whose files carry + # no episode grammar of their own) can show up under a plain numbered + # season folder too, not only one named "Specials" — nothing in + # _title_from_show_siblings should assume otherwise. + show = tmp_path / "Some Show" + season2 = show / "Season 2" + season2.mkdir(parents=True) + (season2 / "One-Off Episode Title.mkv").touch() + season1 = show / "Season 1" + season1.mkdir() + (season1 / "Some Show.S01E01.mkv").touch() + + assert _title_from_show_siblings(season2 / "One-Off Episode Title.mkv") == "Some Show" + + +def test_synthetic_episode_number_is_stable_alphabetical_rank(tmp_path): + folder = tmp_path / "Specials" + folder.mkdir() + (folder / "Bonus Feature One.mkv").touch() + (folder / "Bonus Feature Three.mkv").touch() + (folder / "Bonus Feature Two.mkv").touch() + + assert _synthetic_episode_number(folder / "Bonus Feature One.mkv") == 1 + assert _synthetic_episode_number(folder / "Bonus Feature Three.mkv") == 2 + assert _synthetic_episode_number(folder / "Bonus Feature Two.mkv") == 3 + + # ── end-to-end against a real (tiny, synthetic) video file ────────────────── pytestmark_ffmpeg = pytest.mark.skipif(not _HAVE_FFMPEG, reason="ffmpeg/ffprobe not installed") @@ -201,3 +265,36 @@ async def test_enricher_handles_episode_with_season_from_folder(tmp_path, media_ assert fields["display_title"] == "Some Show" assert fields["season"] == 3 assert fields["episode"] == 7 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_handles_specials_folder_with_no_episode_grammar(tmp_path, media_cache): + # The real bug this guards: every file in a Specials folder is named + # after its own one-off joke, not the show — guessit finds no + # season/episode in any of them, so without the ancestor-folder check + # each one used to fall to the movie branch and get searched against + # TMDB as an unrelated standalone film (found live: a real show's + # Specials folder matched several of its bonus episodes to real, + # unrelated movies that happened to share their one-off titles). + show = tmp_path / "Some Show" + season1 = show / "Season 1" + season1.mkdir(parents=True) + (season1 / "Some Show.S01E01.mkv").touch() + specials = show / "Specials" + specials.mkdir() + (specials / "Bonus Feature One.mkv").touch() + clip = specials / "Bonus Feature Two.mkv" + _make_clip(clip) + entry = IndexEntry(id="fileid3", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + # Not "Bonus Feature Two" — that's this one Special's own title, and + # matching it against TMDB by itself is exactly the bug. The show's + # name, from its ordinary season folder next door. + assert fields["display_title"] == "Some Show" + assert fields["season"] == 0 + assert fields["episode"] is not None -- cgit v1.2.3 From 84aa73e89cfe486c62317f761dcf87d7845eb4f7 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 15:36:23 +0200 Subject: fix(video): TMDB match cache never invalidated on movie<->show reclassification MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The real reason a node restart alone didn't fix already-indexed entries after the previous commit's enrichment change: _do_media_meta_request checked media_cache's file->tmdb mapping (keyed by content hash) and trusted it unconditionally, before ever comparing it against the file's *current* movie/show classification. A file whose season/episode changed on a later scan — exactly what the Specials-folder fix does, for every file it reclassifies from "movie" to "tv" — kept answering with its stale, wrong-kind-of-match forever, since nothing about a reclassification touches this cache or its key. Now falls through to a fresh search whenever the cached media_type disagrees with what the entry resolves to right now, rather than trusting a mapping that predates the file's current classification. --- .../src/meshbay_node/transport/webrtc_server.py | 16 ++++++++-- .../meshbay-node/tests/test_media_meta_request.py | 37 ++++++++++++++++++++++ 2 files changed, 51 insertions(+), 2 deletions(-) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py index 724527b..2b3e3ee 100644 --- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py +++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py @@ -2914,8 +2914,20 @@ class WebRTCPeerSession: meta = None tmdb_id = None if cached is not None: - tmdb_id, media_type = cached - meta = await media_cache.get_tmdb_meta(tmdb_id, media_type) + cached_tmdb_id, cached_media_type = cached + # Trustworthy only if it still agrees with what this file + # resolves to *now*. season/episode come from index-time + # enrichment (enrich.py), which can reclassify a file between + # movie and show on a later scan without this cache knowing — + # it is keyed by the file's content hash alone, which a + # reclassification never changes. Found live: an enrichment fix + # to a Specials-folder bug reclassified hundreds of files from + # "movie" to "tv", and every one kept answering with its + # stale movie-era match forever, because this was trusted + # before ever comparing media_type against the current one. + if cached_media_type == media_type: + tmdb_id = cached_tmdb_id + meta = await media_cache.get_tmdb_meta(tmdb_id, media_type) if meta is None: result, ratio = await self._tmdb_search(tmdb_client, entry, is_show) diff --git a/packages/meshbay-node/tests/test_media_meta_request.py b/packages/meshbay-node/tests/test_media_meta_request.py index a7355e8..a752825 100644 --- a/packages/meshbay-node/tests/test_media_meta_request.py +++ b/packages/meshbay-node/tests/test_media_meta_request.py @@ -112,6 +112,43 @@ async def test_two_episodes_in_the_same_season_folder_each_get_their_own_metadat "two different shows sharing a season folder must not resolve to the same match") +async def test_a_reclassified_file_ignores_its_stale_cached_match(media_cache): + """ + A file's classification (movie vs show) comes from its IndexEntry's + season/episode — set by index-time enrichment, which can change its + mind on a later scan (a filename-parsing fix reclassifying a whole + folder from "movie" to "tv", say) without media_cache's file->tmdb + mapping knowing anything happened: that cache is keyed by the file's + content hash alone, unchanged by any such reclassification. Found + live: exactly this scenario left every affected file answering with + its stale, wrong-kind-of-match forever, since the cache was trusted + before ever comparing media_type against what the entry resolves to + now. + """ + index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) + entry = IndexEntry(id="id-1", name="ep.mkv", path="shows/Show/Specials", + size=1, type="video", added_at=0, + display_title="Some Show", season=0, episode=1) + index.add_entry(entry) + # Pre-populate the cache exactly as it would be left over from before + # entry.season/episode existed — a movie search matched to some + # unrelated title, cached by this file's content hash. + await media_cache.set_file_tmdb("id-1", "stale-movie-id", "movie") + await media_cache.set_tmdb_meta("stale-movie-id", "movie", { + "title": "An Unrelated Movie", "original_title": "An Unrelated Movie", + "release_date": "1999-01-01", "confidence": 1.0, + }) + client = FakeTmdbClient() + session = _session(index, media_cache, client) + + await session._do_media_meta_request({"file_id": "id-1"}) + + resp = session.sent[0] + assert resp["title"] == "Some Show" + assert ("tv", "Some Show") in client.searched + assert resp["tmdb_id"] != "stale-movie-id" + + async def test_missing_file_id_is_refused(media_cache): index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) session = _session(index, media_cache, FakeTmdbClient()) -- cgit v1.2.3 From b1543faad87fd865f8746711ab5f10ac4f331e8b Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 15:51:10 +0200 Subject: fix(video): group every episode under the show's own folder, not a per-file guessit title MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The mechanism itself was wrong, not just the TMDB matching: a bare episode numbering convention with no show name in the filename at all (`001 Episode's Own Title.ext`, no SxxExx, no show prefix — entirely ordinary on its own) makes guessit invent a "title" from whatever text follows the number. That text is the individual episode's own name, and differs for every episode in the folder — trusting it, as the code did, groups nothing together at all: every episode became its own single- episode "show", searched against TMDB by that one-off title alone. Whenever a season-like ancestor folder exists (a numbered season, or Specials/Bonus/Extras -> season 0), its own root folder now names the show unconditionally — never a per-file guessit title, which cannot tell a show's real name from an individual episode's one-off name when the filename carries no reliable ShowName/SxxExx structure. The walk continues past *every* consecutive season-like ancestor, not just the first: a per-season Bonus folder (Show/Season N/Bonus/file.ext) is nested two levels inside the show, both "Bonus" and "Season N" season-like on their own, and stopping at the first would hand back "Season N" as the show's name instead of "Show". Also adds "livre" ("book") to the season-word vocabulary (§3.4) — some shows number their seasons that way (Roman numerals) rather than "season"/"saison". Supersedes _title_from_show_siblings from the previous commit (removed): that fallback assumed a per-file title could still be trusted often enough to be worth borrowing from a sibling season folder — this fix means it never needs trusting in the first place once a season-like ancestor exists. Verified directly against the real library this was found on, not just synthetic tests: every sample file (a numbered episode, a Book/Bonus feature, a Book-root "making of") now resolves to the show's real name. --- .../src/meshbay_node/indexer/enrich.py | 143 ++++++++++----------- .../src/meshbay_node/indexer/title_parse.py | 6 +- packages/meshbay-node/tests/test_enrich.py | 113 +++++++++------- packages/meshbay-node/tests/test_title_parse.py | 7 + 4 files changed, 146 insertions(+), 123 deletions(-) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index fc4b059..5535674 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -38,15 +38,45 @@ MAX_ANCESTOR_DEPTH = 4 MAX_SIBLINGS_CHECKED = 20 -def _season_from_ancestors(file_path: Path) -> int | None: +def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, str] | None: + """ + §3.4/§3.4c: walk up ancestor folders for a season-like one (a numbered + season, or Specials/Bonus/Extras -> season 0) — season from the first + (innermost) match, but the show's own name from *above every + consecutive season-like ancestor*, not just the first one. A + per-season Bonus folder (`Show/Season N/Bonus/file.ext`) is nested two + levels inside the show, both of them season-like on their own + ("Bonus" and "Season N") — stopping at the first would hand back + "Season N" as the show's name instead of "Show". + + Trusted over any per-file guessit title once found: a bare episode + numbering convention with no show name embedded at all + (`001 Episode's Own Title.mkv`, no SxxExx, no show prefix) is + completely ordinary and makes guessit invent a title from whatever + text follows the number — that text is the individual episode's own + name, never the show's, and every episode in the folder produces a + different one. None of that ambiguity exists for the folder structure + itself: the show's own root folder names it once, not per file. + + None if no ancestor looks like a season folder at all — a flat + library has nothing to borrow a show name from this way, and + _title_from_siblings' per-file logic is what applies instead. + """ folder = file_path.parent - for _ in range(MAX_ANCESTOR_DEPTH): - if folder is None or folder == folder.parent: - break - season = title_parse.season_from_folder_name(folder.name) - if season is not None: - return season + season: int | None = None + depth = 0 + while folder is not None and folder.parent != folder and depth < MAX_ANCESTOR_DEPTH: + this_season = title_parse.season_from_folder_name(folder.name) + if this_season is None: + if season is None: + folder = folder.parent + depth += 1 + continue + return season, folder.name + if season is None: + season = this_season folder = folder.parent + depth += 1 return None @@ -82,48 +112,6 @@ def _title_from_siblings(file_path: Path) -> str | None: return None -def _title_from_show_siblings(file_path: Path) -> str | None: - """ - §3.4b: a Specials/Bonus folder's files are routinely named after their - own one-off joke or theme rather than the show at all — every one of - them parses as a standalone "movie" (guessit has no season/episode - grammar to find), and by coincidence a good few of those one-off - titles collide with real, unrelated films: found live, a real show's - Specials folder matched several bonus episodes to real, unrelated - movies sharing those one-off titles, instead of anything to do with - the show. _title_from_siblings can't help — every sibling *here* has the same - problem. What the ordinary season folders next to this one hold does - not: `Show.S01E01.mkv`-shaped filenames name the show properly, so - borrowing from the first one found there is the same trick as - _title_from_siblings, aimed one level higher — at the show's other - seasons rather than this file's own folder. - """ - season_folder = file_path.parent - show_folder = season_folder.parent - try: - sibling_folders = sorted( - (d for d in show_folder.iterdir() if d.is_dir() and d != season_folder), - key=lambda d: d.name) - except OSError: - return None - for folder in sibling_folders: - try: - names = sorted(p.name for p in folder.iterdir() if p.is_file()) - except OSError: - continue - checked = 0 - for name in names: - if Path(name).suffix.lower() not in MEDIA_EXTENSIONS["video"]: - continue - checked += 1 - if checked > MAX_SIBLINGS_CHECKED: - break - parsed = title_parse.parse_episode_filename(name) - if parsed.display_title and parsed.episode is not None: - return parsed.display_title - return None - - def _synthetic_episode_number(file_path: Path) -> int: """ §3.4b: a Specials/Bonus folder's files often carry no episode number at @@ -237,37 +225,38 @@ class Enricher: entry.id, fields.get("duration"), fields.get("width"), fields.get("height")) ep = title_parse.parse_episode_filename(entry.name) - season = ep.season - if season is None: - season = await asyncio.to_thread(_season_from_ancestors, file_path) + ancestor = await asyncio.to_thread(_season_and_show_from_ancestors, file_path) - # §3.4b: an ancestor folder saying this is part of a show (a - # numbered season, or Specials/Bonus/Extras -> season 0) is - # trusted over the filename having no SxxExx of its own — - # otherwise a Specials folder's one-off-named files (no episode - # grammar for guessit to find at all) fall to the movie branch - # below and get searched against TMDB as unrelated standalone - # films, one per Special. Found live: a real show's Specials - # folder matched several bonus episodes to real, unrelated - # movies sharing those one-off titles. - if ep.episode is not None or season is not None: - if ep.episode is not None: - title = ep.display_title or await asyncio.to_thread( - _title_from_siblings, file_path) - episode = ep.episode - else: - # This file's own name carries no episode grammar at - # all — every sibling in *this* folder has the same gap - # (that is what makes it a Specials-shaped folder), so - # borrowing from them (_title_from_siblings) would just - # hand back another Special's own one-off title. The - # show's ordinary season folders, next to this one, - # don't have that gap. - title = await asyncio.to_thread(_title_from_show_siblings, file_path) + if ancestor is not None: + # A season-like ancestor folder exists — the show's own + # root folder names it, trusted over any per-file guessit + # title. This is what actually groups every episode under + # one show: a bare numbering convention with no show name + # in the filename at all (`001 Episode's Own Title.mkv`, + # ordinary enough on its own) makes guessit invent a title + # from whatever text follows the number, which is that + # episode's own name, never the show's — and differs for + # every episode, so nothing would ever group together. + # Found live: a real show's episodes and its Specials + # folder alike, both named this way, showed up as + # individual "movies" each searched against — and matched + # to — an unrelated real film sharing that one-off title, + # instead of anything grouped under the show at all. + season, show_title = ancestor + episode = ep.episode + if episode is None: episode = await asyncio.to_thread(_synthetic_episode_number, file_path) - fields["display_title"] = title or title_parse.naive_title(entry.name) - fields["season"] = season if season is not None else 0 + fields["display_title"] = show_title + fields["season"] = season fields["episode"] = episode + elif ep.episode is not None: + # No season-like ancestor at all (a flat library) but the + # filename itself carries season+episode (§3.4). + title = ep.display_title or await asyncio.to_thread( + _title_from_siblings, file_path) + fields["display_title"] = title or title_parse.naive_title(entry.name) + fields["season"] = ep.season + fields["episode"] = ep.episode else: mv = title_parse.parse_movie_filename(entry.name) fields["display_title"] = mv.display_title or mv.naive_title diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 5203247..0baf34f 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -36,8 +36,10 @@ _EDITION_RE = re.compile("|".join(_EDITION_PHRASES), re.IGNORECASE) # A season-like ancestor folder: the English/French words plus a number or # Roman numeral. Vocabulary is a plain tuple so a deployment can extend it -# per locale without touching the regex-building logic. -SEASON_WORDS = ("season", "saison") +# per locale without touching the regex-building logic. "livre" ("book") is +# real, observed vocabulary too — some shows name their seasons that way +# (Roman numerals: "Livre I".."Livre VI") rather than "saison". +SEASON_WORDS = ("season", "saison", "livre") _SEASON_RE = re.compile( r"(?:" + "|".join(SEASON_WORDS) + r")\s*([0-9]+|[ivxlc]+)\b", re.IGNORECASE, diff --git a/packages/meshbay-node/tests/test_enrich.py b/packages/meshbay-node/tests/test_enrich.py index 5497812..8c0c6b2 100644 --- a/packages/meshbay-node/tests/test_enrich.py +++ b/packages/meshbay-node/tests/test_enrich.py @@ -9,8 +9,8 @@ import pytest from meshbay_common.protocol import IndexEntry from meshbay_node.indexer.enrich import ( - Enricher, _season_from_ancestors, _synthetic_episode_number, - _title_from_show_siblings, _title_from_siblings, + Enricher, _season_and_show_from_ancestors, _synthetic_episode_number, + _title_from_siblings, ) from meshbay_node.media_cache import MediaCache @@ -19,22 +19,48 @@ _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") # ── pure helpers, no ffmpeg needed ─────────────────────────────────────────── -def test_season_from_ancestors_finds_season_folder(tmp_path): +def test_season_and_show_from_ancestors_finds_season_folder(tmp_path): folder = tmp_path / "Some Show" / "Season 2" folder.mkdir(parents=True) ep = folder / "01 - Episode Title.mkv" ep.touch() - assert _season_from_ancestors(ep) == 2 + assert _season_and_show_from_ancestors(ep) == (2, "Some Show") -def test_season_from_ancestors_none_when_no_season_folder(tmp_path): +def test_season_and_show_from_ancestors_none_when_no_season_folder(tmp_path): folder = tmp_path / "Movies" folder.mkdir() f = folder / "Some Movie 2015.mkv" f.touch() - assert _season_from_ancestors(f) is None + assert _season_and_show_from_ancestors(f) is None + + +def test_season_and_show_from_ancestors_walks_past_a_per_season_bonus_folder(tmp_path): + # The real shape this guards: a Specials/Bonus folder nested *inside* + # each numbered season folder (Show/Season N/Bonus/file.ext) rather + # than once at the show's top level — both "Bonus" and "Season N" are + # season-like on their own, and stopping at the first (innermost) + # would hand back "Season N" as the show's name instead of "Show". + folder = tmp_path / "Some Show" / "Season 1" / "Bonus" + folder.mkdir(parents=True) + f = folder / "One-Off Bonus Title.mkv" + f.touch() + + assert _season_and_show_from_ancestors(f) == (0, "Some Show") + + +def test_season_and_show_from_ancestors_book_word_season_folder(tmp_path): + # Same shape as the real bug this whole fix guards, with the "book" + # season vocabulary (§3.4) instead of "season"/"saison": a plain + # numbered-episode folder nested under it. + folder = tmp_path / "Some Show" / "Livre I" / "Episodes" + folder.mkdir(parents=True) + f = folder / "001 Episode's Own Title.mkv" + f.touch() + + assert _season_and_show_from_ancestors(f) == (1, "Some Show") def test_title_from_siblings_borrows_from_a_titled_sibling(tmp_path): @@ -67,44 +93,6 @@ def test_title_from_siblings_ignores_a_titled_sibling_with_no_episode_number(tmp assert _title_from_siblings(folder / "Bonus Feature Two.mkv") is None -def test_title_from_show_siblings_borrows_from_a_sibling_season_folder(tmp_path): - show = tmp_path / "Some Show" - specials = show / "Specials" - specials.mkdir(parents=True) - (specials / "Bonus Feature One.mkv").touch() - season1 = show / "Season 1" - season1.mkdir() - (season1 / "Some Show.S01E01.mkv").touch() - - assert _title_from_show_siblings(specials / "Bonus Feature One.mkv") == "Some Show" - - -def test_title_from_show_siblings_none_when_no_show_folder_has_one(tmp_path): - show = tmp_path / "Some Show" - specials = show / "Specials" - specials.mkdir(parents=True) - (specials / "Bonus Feature One.mkv").touch() - (specials / "Bonus Feature Two.mkv").touch() - - assert _title_from_show_siblings(specials / "Bonus Feature One.mkv") is None - - -def test_title_from_show_siblings_not_specific_to_the_word_specials(tmp_path): - # Genericity check: the same gap (a season-like folder whose files carry - # no episode grammar of their own) can show up under a plain numbered - # season folder too, not only one named "Specials" — nothing in - # _title_from_show_siblings should assume otherwise. - show = tmp_path / "Some Show" - season2 = show / "Season 2" - season2.mkdir(parents=True) - (season2 / "One-Off Episode Title.mkv").touch() - season1 = show / "Season 1" - season1.mkdir() - (season1 / "Some Show.S01E01.mkv").touch() - - assert _title_from_show_siblings(season2 / "One-Off Episode Title.mkv") == "Some Show" - - def test_synthetic_episode_number_is_stable_alphabetical_rank(tmp_path): folder = tmp_path / "Specials" folder.mkdir() @@ -298,3 +286,40 @@ async def test_enricher_handles_specials_folder_with_no_episode_grammar(tmp_path assert fields["display_title"] == "Some Show" assert fields["season"] == 0 assert fields["episode"] is not None + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_groups_bare_numbered_episodes_under_the_show_folder(tmp_path, media_cache): + """ + The real-world shape this whole fix is for: every episode (and every + Bonus feature) named " ", with the + show's name appearing nowhere in any filename at all — only in the + show's own root folder. guessit still finds an episode number here + (unlike the plain-Specials case above), and invents a "title" from + whatever text follows it — a different one for every file. Trusting + that per-file title, as the code used to, groups nothing together at + all: every episode becomes its own single-episode "show", searched + against TMDB by its own one-off title, and mismatched to whichever + unrelated real film or show happens to share it. + """ + show = tmp_path / "Some Show" + season1_eps = show / "Livre I" / "Episodes" + season1_eps.mkdir(parents=True) + (season1_eps / "001 First Episode's Own Title.mkv").touch() + clip = season1_eps / "002 Second Episode's Own Title.mkv" + _make_clip(clip) + season2_eps = show / "Livre II" / "Episodes" + season2_eps.mkdir(parents=True) + (season2_eps / "001 A Season 2 Episode's Own Title.mkv").touch() + entry = IndexEntry(id="fileid4", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + # Not "Second Episode's Own Title" — every episode has a different + # one, and none of them is the show. + assert fields["display_title"] == "Some Show" + assert fields["season"] == 1 + assert fields["episode"] == 2 diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index 1a277f2..1881613 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -111,6 +111,13 @@ def test_season_folder_roman_numeral(): assert season_from_folder_name("Saison IV") == 4 +def test_season_folder_book_word_roman_numeral(): + # Some shows number their seasons by "book" (Roman numerals) rather + # than "season"/"saison" — real, observed vocabulary (§3.4). + assert season_from_folder_name("Livre I") == 1 + assert season_from_folder_name("Livre VI") == 6 + + def test_specials_folder_maps_to_season_zero(): assert season_from_folder_name("Specials") == 0 assert season_from_folder_name("Bonus") == 0 -- cgit v1.2.3 From c81eaa0bd07e360ec03406dfc63267c3013a0319 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 16:04:00 +0200 Subject: fix(video): fix two remaining bugs in Specials numbering and long-season episodes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1. A season spanning more than one folder (a per-book Bonus folder nested inside every numbered season) got the same synthetic episode numbers handed out again in each folder independently — three unrelated Specials all showing up as "S0E01". _synthetic_episode_number now ranks across the whole show for the target season, not just one file's own folder. 2. guessit reads a bare 3-digit leading episode number as a concatenated season+episode guess rather than a plain episode number — "100" parses as season=1, episode=0, not episode=100, with nothing in its output distinguishing that from a real 2-digit episode. A season-like ancestor already overrides guessit's season (previous commit); this applies the same fix to episode via a direct regex on the leading number, capped at 3 digits so a leading year (4 digits) is never misread the same way. Both confirmed against the real library this whole fix was found on: per-book Bonus features across two books no longer collide, and a real 100th-episode file now resolves to episode 100 instead of 0. --- .../src/meshbay_node/indexer/enrich.py | 78 ++++++++++++++++------ .../src/meshbay_node/indexer/title_parse.py | 17 +++++ packages/meshbay-node/tests/test_enrich.py | 72 +++++++++++++++++--- packages/meshbay-node/tests/test_title_parse.py | 25 +++++++ 4 files changed, 159 insertions(+), 33 deletions(-) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index 5535674..07f0be7 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -36,9 +36,14 @@ MAX_ANCESTOR_DEPTH = 4 # Bounds the "borrow a title from a sibling episode filename" scan (§3.4) so # a folder with thousands of files costs a fixed, small amount of work. MAX_SIBLINGS_CHECKED = 20 +# Bounds the whole-show scan _synthetic_episode_number uses to rank a +# season's files across more than one folder (§3.4c) — a show's total file +# count, not just one folder's, so this needs more headroom than +# MAX_SIBLINGS_CHECKED. +MAX_SEASON_FILES_CHECKED = 500 -def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, str] | None: +def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, Path] | None: """ §3.4/§3.4c: walk up ancestor folders for a season-like one (a numbered season, or Specials/Bonus/Extras -> season 0) — season from the first @@ -72,7 +77,7 @@ def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, str] | None: folder = folder.parent depth += 1 continue - return season, folder.name + return season, folder if season is None: season = this_season folder = folder.parent @@ -112,27 +117,45 @@ def _title_from_siblings(file_path: Path) -> str | None: return None -def _synthetic_episode_number(file_path: Path) -> int: +def _synthetic_episode_number(file_path: Path, show_root: Path, season: int) -> int: """ - §3.4b: a Specials/Bonus folder's files often carry no episode number at - all — each is just named after its own one-off title. The frontend - (video-app.js's buildSeasons) sorts within a season by this number but - only needs it to provide a stable order, not to mean anything beyond - that — alphabetical rank among the folder's video files is enough, and - deterministic across re-scans as long as the folder's contents don't - change. 1-based so it reads as "episode 1", not "episode 0", in a UI - that already uses season 0 for "Specials" itself. + §3.4b/§3.4c: a Specials/Bonus folder's files often carry no episode + number at all — each is just named after its own one-off title. The + frontend (video-app.js's buildSeasons) sorts within a season by this + number but only needs it to provide a stable order, not to mean + anything beyond that. + + Ranked across the *whole show*, not just this file's own folder: season + 0 routinely spans more than one folder under the show's root — a + Bonus folder nested inside every numbered season + (`Show/Season N/Bonus/...`) is exactly the shape that produces this — + and ranking within just this file's own folder hands out the same + "episode 1" again in every other one, found live as three unrelated + Specials all showing up as "S0E01". Deterministic across re-scans as + long as the show's contents don't change. 1-based so it reads as + "episode 1", not "episode 0", in a UI that already uses season 0 for + "Specials" itself. Bounded (MAX_SEASON_FILES_CHECKED) so a huge show + costs a fixed amount of work rather than scaling with its size. """ try: - names = sorted( - p.name for p in file_path.parent.iterdir() - if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]) + candidates = sorted( + (p for p in show_root.rglob("*") + if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]), + key=lambda p: (str(p.parent), p.name), + ) except OSError: return 1 - try: - return names.index(file_path.name) + 1 - except ValueError: - return 1 + rank = 0 + for i, candidate in enumerate(candidates): + if i >= MAX_SEASON_FILES_CHECKED: + break + ancestor = _season_and_show_from_ancestors(candidate) + if ancestor is None or ancestor[0] != season: + continue + rank += 1 + if candidate == file_path: + return rank + return 1 async def _make_thumbnail(file_path: Path, duration: float | None) -> bytes | None: @@ -242,11 +265,22 @@ class Enricher: # individual "movies" each searched against — and matched # to — an unrelated real film sharing that one-off title, # instead of anything grouped under the show at all. - season, show_title = ancestor - episode = ep.episode + season, show_folder = ancestor + # Not ep.episode: guessit reads a bare 3-digit leading + # number (routine once a season-like ancestor is already + # doing the real season/episode work) as a concatenated + # season+episode guess rather than a plain episode number + # — confirmed live, "100 Title.mkv" parsed as episode 0, + # not 100, with nothing in guessit's own output telling + # that apart from a real 2-digit episode. The direct regex + # reads the whole leading number as it is. + episode = title_parse.leading_episode_number(entry.name) + if episode is None: + episode = ep.episode if episode is None: - episode = await asyncio.to_thread(_synthetic_episode_number, file_path) - fields["display_title"] = show_title + episode = await asyncio.to_thread( + _synthetic_episode_number, file_path, show_folder, season) + fields["display_title"] = show_folder.name fields["season"] = season fields["episode"] = episode elif ep.episode is not None: diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 0baf34f..985bd7a 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -229,3 +229,20 @@ def parse_episode_filename(filename: str) -> ParsedName: display_title=title or None, naive_title=nt, season=season, episode=episode, confidence=confidence, ) + + +# A bare leading episode number, no show name attached (§3.4c) — the same +# shape as music's _TRACK_PREFIX_RE, capped at 3 digits for the same reason: +# a leading year ("2010 - Episode.mkv") is 4 digits and must not match. +# guessit's own `episode` is not a substitute here: given exactly 3 digits it +# tries to read them as a concatenated SxxE/SEE season+episode pair instead +# of a plain episode number — confirmed live, "100 Title.mkv" parses as +# season=1, episode=0, not episode=100 — silently wrong in a way nothing +# about its output distinguishes from a real 2-digit episode. This reads the +# whole leading number as one value instead. +_LEADING_NUMBER_RE = re.compile(r"^(\d{1,3})[\s._-]+(?=\S)") + + +def leading_episode_number(filename: str) -> int | None: + m = _LEADING_NUMBER_RE.match(filename) + return int(m.group(1)) if m else None diff --git a/packages/meshbay-node/tests/test_enrich.py b/packages/meshbay-node/tests/test_enrich.py index 8c0c6b2..4b957cd 100644 --- a/packages/meshbay-node/tests/test_enrich.py +++ b/packages/meshbay-node/tests/test_enrich.py @@ -20,12 +20,13 @@ _HAVE_FFMPEG = shutil.which("ffmpeg") and shutil.which("ffprobe") # ── pure helpers, no ffmpeg needed ─────────────────────────────────────────── def test_season_and_show_from_ancestors_finds_season_folder(tmp_path): - folder = tmp_path / "Some Show" / "Season 2" + show = tmp_path / "Some Show" + folder = show / "Season 2" folder.mkdir(parents=True) ep = folder / "01 - Episode Title.mkv" ep.touch() - assert _season_and_show_from_ancestors(ep) == (2, "Some Show") + assert _season_and_show_from_ancestors(ep) == (2, show) def test_season_and_show_from_ancestors_none_when_no_season_folder(tmp_path): @@ -43,24 +44,26 @@ def test_season_and_show_from_ancestors_walks_past_a_per_season_bonus_folder(tmp # than once at the show's top level — both "Bonus" and "Season N" are # season-like on their own, and stopping at the first (innermost) # would hand back "Season N" as the show's name instead of "Show". - folder = tmp_path / "Some Show" / "Season 1" / "Bonus" + show = tmp_path / "Some Show" + folder = show / "Season 1" / "Bonus" folder.mkdir(parents=True) f = folder / "One-Off Bonus Title.mkv" f.touch() - assert _season_and_show_from_ancestors(f) == (0, "Some Show") + assert _season_and_show_from_ancestors(f) == (0, show) def test_season_and_show_from_ancestors_book_word_season_folder(tmp_path): # Same shape as the real bug this whole fix guards, with the "book" # season vocabulary (§3.4) instead of "season"/"saison": a plain # numbered-episode folder nested under it. - folder = tmp_path / "Some Show" / "Livre I" / "Episodes" + show = tmp_path / "Some Show" + folder = show / "Livre I" / "Episodes" folder.mkdir(parents=True) f = folder / "001 Episode's Own Title.mkv" f.touch() - assert _season_and_show_from_ancestors(f) == (1, "Some Show") + assert _season_and_show_from_ancestors(f) == (1, show) def test_title_from_siblings_borrows_from_a_titled_sibling(tmp_path): @@ -94,15 +97,36 @@ def test_title_from_siblings_ignores_a_titled_sibling_with_no_episode_number(tmp def test_synthetic_episode_number_is_stable_alphabetical_rank(tmp_path): - folder = tmp_path / "Specials" - folder.mkdir() + show = tmp_path / "Some Show" + folder = show / "Specials" + folder.mkdir(parents=True) (folder / "Bonus Feature One.mkv").touch() (folder / "Bonus Feature Three.mkv").touch() (folder / "Bonus Feature Two.mkv").touch() - assert _synthetic_episode_number(folder / "Bonus Feature One.mkv") == 1 - assert _synthetic_episode_number(folder / "Bonus Feature Three.mkv") == 2 - assert _synthetic_episode_number(folder / "Bonus Feature Two.mkv") == 3 + assert _synthetic_episode_number(folder / "Bonus Feature One.mkv", show, 0) == 1 + assert _synthetic_episode_number(folder / "Bonus Feature Three.mkv", show, 0) == 2 + assert _synthetic_episode_number(folder / "Bonus Feature Two.mkv", show, 0) == 3 + + +def test_synthetic_episode_number_does_not_collide_across_per_season_bonus_folders(tmp_path): + # The real bug this guards: season 0 spanning more than one folder + # (a Bonus folder nested inside each numbered season) — ranking within + # just one file's own folder hands out "episode 1" again in every + # other one, found live as three unrelated Specials all showing up as + # "S0E01". Ranked across the whole show instead, so each gets its own + # number. + show = tmp_path / "Some Show" + bonus1 = show / "Season 1" / "Bonus" + bonus1.mkdir(parents=True) + (bonus1 / "First Season's Bonus.mkv").touch() + bonus2 = show / "Season 2" / "Bonus" + bonus2.mkdir(parents=True) + (bonus2 / "Second Season's Bonus.mkv").touch() + + n1 = _synthetic_episode_number(bonus1 / "First Season's Bonus.mkv", show, 0) + n2 = _synthetic_episode_number(bonus2 / "Second Season's Bonus.mkv", show, 0) + assert n1 != n2 # ── end-to-end against a real (tiny, synthetic) video file ────────────────── @@ -323,3 +347,29 @@ async def test_enricher_groups_bare_numbered_episodes_under_the_show_folder(tmp_ assert fields["display_title"] == "Some Show" assert fields["season"] == 1 assert fields["episode"] == 2 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_enricher_reads_a_three_digit_episode_number_correctly(tmp_path, media_cache): + """ + guessit's own episode number is not trustworthy here: given a bare + leading "100", it reads that as a concatenated season+episode guess + (season=1, episode=0) rather than episode 100 — confirmed live, and + indistinguishable from a real 2-digit episode in its output. The + season-like ancestor already overrides guessit's season; this is the + same fix applied to episode. + """ + show = tmp_path / "Some Show" + season_eps = show / "Season 6" / "Episodes" + season_eps.mkdir(parents=True) + clip = season_eps / "100 A Long Season's Own Title.mkv" + _make_clip(clip) + entry = IndexEntry(id="fileid5", name=clip.name, path=str(clip.relative_to(tmp_path)), + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + assert fields["season"] == 6 + assert fields["episode"] == 100 diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index 1881613..cf51646 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -6,6 +6,7 @@ is a manual acceptance step (§11), not something this repo's corpus holds. from meshbay_node.indexer.title_parse import ( ParsedName, + leading_episode_number, naive_title, parse_episode_filename, parse_movie_filename, @@ -128,6 +129,30 @@ def test_non_season_folder_name_returns_none(): assert season_from_folder_name("Some Show Name") is None +# ── §3.4c: a bare leading episode number, guessit's 3-digit blind spot ─────── + +def test_leading_episode_number_reads_the_whole_number(): + assert leading_episode_number("001 Episode's Own Title.mkv") == 1 + assert leading_episode_number("099 Episode's Own Title.mkv") == 99 + + +def test_leading_episode_number_not_split_by_guessit_at_three_digits(): + # The real bug this guards: guessit itself reads a bare "100" as season=1, + # episode=0 (a concatenated SxxE guess), not episode=100 — silently, with + # nothing in its output telling apart a real 2-digit episode from this. + assert leading_episode_number("100 Episode's Own Title.mkv") == 100 + + +def test_leading_episode_number_ignores_a_leading_year(): + # Four digits, not the 1-3 an episode-number prefix can be — this is a + # normal movie filename shape, not an episode number. + assert leading_episode_number("2010 - Some Movie.mkv") is None + + +def test_leading_episode_number_none_without_a_leading_number(): + assert leading_episode_number("Some Show.S01E01.mkv") is None + + def test_parsed_name_is_a_plain_dataclass(): # sanity: constructible with just the one required field, per the # "None => caller must supply from elsewhere" contract. -- cgit v1.2.3 From bc94dc672d142e989fa4f483f643d4b43f66b02f Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 16:24:18 +0200 Subject: fix(video): recognize "S1"/"S2"-style season folders, not just full words MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found live: a real show organized its first season as "House.of.the.Dragon. S01E01...mkv" inside a folder named "S1" (the show name embedded in every filename, so grouping worked by guessit's own title alone), but its later seasons as "S02E01. Episode's Own Title.mkv" inside "S2"/"S3" — no show name in any filename at all, relying entirely on the folder. Those seasons showed up as loose individual entries instead of grouped under the show: season_from_folder_name only recognized "season"/"saison"/"livre" as full words, so "S2" didn't register as a season folder at all, and the ancestor-based show-grouping fix (previous commits) never triggered for those seasons. A bare "S" + 1-2 digits as the *whole* folder name is now recognized too — anchored to the entire name so it can't match some unrelated folder that merely starts with "s" followed by digits. --- .../src/meshbay_node/indexer/title_parse.py | 9 +++++++++ packages/meshbay-node/tests/test_title_parse.py | 18 ++++++++++++++++++ 2 files changed, 27 insertions(+) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 985bd7a..24b423e 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -45,6 +45,12 @@ _SEASON_RE = re.compile( re.IGNORECASE, ) _SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE) +# A bare "S" + number as the *whole* folder name — "S1", "S2", "S02" — a +# common abbreviated convention distinct from SEASON_WORDS' full words. +# Anchored to the entire name, not just `\b`-bounded within a longer +# string, so it only matches a folder actually named just that — never +# some other word that merely starts with "s" followed by digits. +_SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE) _ROMAN_NUMERALS = { 2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI", @@ -110,6 +116,9 @@ def season_from_folder_name(name: str) -> int | None: """ if _SPECIALS_RE.search(name): return 0 + m = _SEASON_ABBREV_RE.match(name.strip()) + if m: + return int(m.group(1)) m = _SEASON_RE.search(name) if not m: return None diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index cf51646..5f0977c 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -119,6 +119,24 @@ def test_season_folder_book_word_roman_numeral(): assert season_from_folder_name("Livre VI") == 6 +def test_season_folder_bare_s_abbreviation(): + # A common abbreviated convention distinct from SEASON_WORDS' full + # words — found live: a show with "S1"/"S2"/"S3" folders instead of + # "Season 1" etc, exactly the shape that grouped its later seasons as + # loose individual entries rather than under the show. + assert season_from_folder_name("S1") == 1 + assert season_from_folder_name("S2") == 2 + assert season_from_folder_name("S02") == 2 + + +def test_season_folder_bare_s_abbreviation_does_not_match_inside_a_longer_name(): + # Anchored to the whole folder name — a real folder that merely starts + # with "S" followed by digits somewhere in a longer, unrelated name + # must not be read as a season abbreviation. + assert season_from_folder_name("S1 Extended Cut") is None + assert season_from_folder_name("Something2") is None + + def test_specials_folder_maps_to_season_zero(): assert season_from_folder_name("Specials") == 0 assert season_from_folder_name("Bonus") == 0 -- cgit v1.2.3 From 59289b82dc8af08c605fd247c90a5c5ec92c6db3 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Wed, 26 Aug 2026 17:09:07 +0200 Subject: fix(video): a TMDB override never stored the metadata its chosen id names MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found live: "fix match" appeared to work for two shows but not for a movie whose own automatic search kept landing on the same wrong result. The override only ever recorded the file->tmdb_id mapping — never the metadata that id actually names. _do_media_meta_request's cache check agrees the mapping is fresh (same media_type) but finds nothing under that *new* id in tmdb_meta, since nothing had ever fetched it, and falls through to a brand-new search using the file's own title — reproducing the exact match the override was meant to replace. This stayed invisible for the two shows only because their own title happened to be enough for that fallback search to land on the right answer anyway, entirely independent of whatever the override recorded — never because the override was actually being honored. It surfaced on a movie whose own title search kept landing on the same wrong match regardless. Fetches and stores the real metadata for the chosen tmdb_id up front (via the existing _tmdb_build_meta, which needs only the id — no search result object required), so a later lookup finds the override itself instead of falling through to a search blind to it. --- .../src/meshbay_node/transport/webrtc_server.py | 18 +++++++ .../tests/test_tmdb_override_policy.py | 58 ++++++++++++++++++++++ 2 files changed, 76 insertions(+) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py index 2b3e3ee..252e202 100644 --- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py +++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py @@ -3117,6 +3117,24 @@ class WebRTCPeerSession: if entry is None or media_cache is None: self._send({"type": "error", "detail": "File or media cache not available"}) return + # This only ever recorded the file->tmdb_id mapping, never the + # metadata tmdb_id names — _do_media_meta_request's cache check + # (entry, cache) both agree on media_type, so it trusted the + # mapping — but found nothing under this *new* id in tmdb_meta + # (nothing had ever fetched it), and silently fell through to a + # fresh search using the file's own title, exactly the one that + # produced the wrong match in the first place. Confirmed live: an + # override "stuck" for shows only because their own title happened + # to be enough for that fallback search to land on the right + # answer anyway, coincidentally — never because the override itself + # was actually being honored — and was invisible until a movie + # whose own title search kept landing on the same wrong result + # exposed it. Fetching and storing the real metadata up front is + # what makes the *override* the thing a later lookup finds. + tmdb_client = self._ctx.get("tmdb_client") + if tmdb_client is not None: + meta = await self._tmdb_build_meta(tmdb_client, tmdb_id, media_type, {}) + await media_cache.set_tmdb_meta(tmdb_id, media_type, meta) target_title = entry.display_title or entry.name matched = [e for e in ctx["index"].entries if e.type == "video" and (e.display_title or e.name) == target_title] diff --git a/packages/meshbay-node/tests/test_tmdb_override_policy.py b/packages/meshbay-node/tests/test_tmdb_override_policy.py index cd1cee1..c2f28e6 100644 --- a/packages/meshbay-node/tests/test_tmdb_override_policy.py +++ b/packages/meshbay-node/tests/test_tmdb_override_policy.py @@ -195,5 +195,63 @@ async def test_override_updates_every_entry_sharing_the_display_title(tmp_path): await media_cache.close() +class _FakeTmdbClient: + """Just enough for _tmdb_build_meta to run end to end — a fixed, + deterministic response, not a search stub (the override already has a + chosen tmdb_id; nothing here should need to search for anything).""" + + async def movie_details(self, tmdb_id, language=None): + return {"title": "The Corrected Title", "overview": "A correct overview.", + "poster_path": "/poster.jpg", "genres": [{"name": "Drama"}]} + + async def tv_details(self, tmdb_id, language=None): + return await self.movie_details(tmdb_id, language) + + async def movie_credits(self, tmdb_id): + return {"cast": [], "crew": []} + + async def tv_credits(self, tmdb_id): + return {"cast": [], "crew": []} + + +async def test_override_stores_the_chosen_matchs_metadata_not_just_its_id(tmp_path): + """ + The real bug this guards: only the file->tmdb_id mapping ever got + recorded, never the metadata the chosen id actually names. + _do_media_meta_request's cache check agrees the mapping is fresh (same + media_type) but finds nothing under that id in tmdb_meta — nothing had + ever fetched it — and falls through to a brand new search using the + file's own title, reproducing the very match the override was meant to + replace. Confirmed live: this stayed invisible for shows whose own + title happened to be enough for that fallback search to land on the + right answer anyway, and surfaced on a movie whose own title kept + landing on the same wrong match regardless of the override. + """ + session = _session(tmp_path, "op", operator="op") + index = session._ctx["index"] + entry = _entry("shared", "movie.mkv", "Some Movie's Own Wrong Title") + index.add_entry(entry) + + media_cache = MediaCache(db_path=tmp_path / "media_cache.db") + await media_cache.open() + try: + session._ctx["media_cache"] = media_cache + session._ctx["tmdb_client"] = _FakeTmdbClient() + session._verify_admin_sig = lambda transcript, sig: _true() + session._peer_registry = lambda: {} + + await session._admin_exec_tmdb_override( + {"subject": f"file_id={entry.id},tmdb_id=999,media_type=movie"}, + b"transcript", b"sig") + + meta = await media_cache.get_tmdb_meta("999", "movie") + assert meta is not None, ( + "the override must store the metadata its chosen id actually names, " + "not just the file->tmdb_id mapping") + assert meta["title"] == "The Corrected Title" + finally: + await media_cache.close() + + async def _true(): return True -- cgit v1.2.3