From 00880fd8b9d362ad391c784f1ee31bd62aa18d62 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Sat, 29 Aug 2026 18:45:37 +0200 Subject: fix(node): stop a movie with a mangled quality tag being shelved as a series MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found on demo35: "Some.Film.2017.MULTI.108.grp.mkv" — the release name's "1080p" truncated to "108" — makes guessit read S01E08, so enrich.py's flat-library branch (elif ep.episode is not None) filed a standalone film as a series. "Fix match" then only offered TV results for the phantom show, so there was no way out from the UI. - title_parse.has_episode_marker(): true only for an explicit SxxExx / 1x08 / "Episode N" / "Season N" token, not a bare 3-4 digit run. - enrich.py: the flat-library branch now needs ep.season AND ep.episode, plus either a real marker or the absence of a "(2019)"-style year. Every genuine flat-dumped episode in the corpus carries a marker, so real shows are untouched; the same misparse on "1280" ("...2013.1280...") is covered too. docs/mediacenter.md §10.2. Known gap left open: no operator control over the movie/show kind itself — a "this is a movie / a show" toggle would. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018BMLQjqFGCize2KtNBT79v --- .../src/meshbay_node/indexer/enrich.py | 13 +++++-- .../src/meshbay_node/indexer/title_parse.py | 20 +++++++++++ packages/meshbay-node/tests/test_enrich.py | 41 ++++++++++++++++++++++ packages/meshbay-node/tests/test_title_parse.py | 22 ++++++++++++ 4 files changed, 94 insertions(+), 2 deletions(-) (limited to 'packages/meshbay-node') diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index 07f0be7..8b2eca5 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -283,9 +283,18 @@ class Enricher: fields["display_title"] = show_folder.name fields["season"] = season fields["episode"] = episode - elif ep.episode is not None: + elif (ep.season is not None and ep.episode is not None + and (title_parse.has_episode_marker(entry.name) + or title_parse.year_in(entry.name) is None)): # No season-like ancestor at all (a flat library) but the - # filename itself carries season+episode (§3.4). + # filename itself carries season+episode (§3.4) — *and* it + # is a real marker, not guessit reading a bare number as + # SxxExx. A movie whose "1080p" tag was truncated to "108", + # or "1280" left in the name, otherwise parses to S01E08 / + # S12E80 and gets shelved as a nonexistent series + # (found live 2026-08-29). A genuine flat-dumped episode + # has an explicit SxxExx/1x08/"Episode N" marker; a movie + # has a "(2019)"-style year and no such marker. title = ep.display_title or await asyncio.to_thread( _title_from_siblings, file_path) fields["display_title"] = title or title_parse.naive_title(entry.name) diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 018746b..def947f 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -288,6 +288,26 @@ def strip_track_prefix(text: str) -> str: return re.sub(r"^[\s-]+", "", text[m.end():]).strip() if m else text +# An *explicit* season/episode marker: SxxExx, 1x08, "Episode 8", "Ep 8", +# "Season 1"/"Saison 1". guessit will also invent a season+episode from a +# bare 3-4 digit run ("1080p" truncated to "108" -> S01E08; "1280" -> +# S12E80), which is how a plain movie ends up shelved as a series +# (§10.1/V14). The indexer uses this to tell a real flat-library episode +# from that hallucination. +_EPISODE_MARKER_RE = re.compile( + r"s\d{1,2}[\s._-]*e\d{1,3}" + r"|\b\d{1,2}x\d{1,3}\b" + r"|\bepisode[\s._-]*\d{1,3}\b" + r"|\bep[\s._-]*\d{1,3}\b" + r"|\b(?:season|saison)[\s._-]*\d{1,2}\b", + re.IGNORECASE, +) + + +def has_episode_marker(filename: str) -> bool: + return bool(_EPISODE_MARKER_RE.search(filename)) + + def parse_episode_filename(filename: str) -> ParsedName: """ Parse an episode filename. `display_title` may come back None (e.g. diff --git a/packages/meshbay-node/tests/test_enrich.py b/packages/meshbay-node/tests/test_enrich.py index 4b957cd..a35c713 100644 --- a/packages/meshbay-node/tests/test_enrich.py +++ b/packages/meshbay-node/tests/test_enrich.py @@ -373,3 +373,44 @@ async def test_enricher_reads_a_three_digit_episode_number_correctly(tmp_path, m assert fields["season"] == 6 assert fields["episode"] == 100 + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_a_movie_with_a_mangled_quality_tag_is_not_shelved_as_a_series( + tmp_path, media_cache): + """ + Found live 2026-08-29: a standalone film whose "1080p" tag was + truncated to "108" in the filename makes guessit invent S01E08, so + enrich (flat library, no season ancestor) filed it as a nonexistent + series. A real flat-dumped episode carries an explicit SxxExx / 1x08 / + "Episode N" marker; a movie has a "(2017)"-style year and none. + """ + clip = tmp_path / "Some.Film.2017.MULTI.108.grp.mkv" + _make_clip(clip) + entry = IndexEntry(id="fid-trunc", name=clip.name, path=clip.name, + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + assert fields.get("season") is None and fields.get("episode") is None, ( + "a movie with a mangled quality tag must not become a series") + assert fields["display_title"] + + +@pytestmark_ffmpeg +@pytest.mark.asyncio +async def test_a_flat_episode_with_a_real_marker_stays_a_show_even_with_a_year( + tmp_path, media_cache): + """The guard must not misfire: a genuine flat-dumped episode that also + carries a year has an explicit SxxExx marker and stays a show.""" + clip = tmp_path / "Some.Show.2022.S01E02.1080p.WEB.mkv" + _make_clip(clip) + entry = IndexEntry(id="fid-marker", name=clip.name, path=clip.name, + size=clip.stat().st_size, type="video", added_at=0) + + enricher = Enricher(media_cache) + _, fields = await _run(enricher, entry, clip) + + assert fields["season"] == 1 and fields["episode"] == 2 diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index ac4d434..045635f 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -7,6 +7,7 @@ is a manual acceptance step (§11), not something this repo's corpus holds. from meshbay_node.indexer.title_parse import ( ParsedName, clean_query, + has_episode_marker, leading_episode_number, naive_title, parse_episode_filename, @@ -113,6 +114,27 @@ def test_clean_query_despaces_a_folder_name_without_eating_the_last_word(): assert naive_title("Some.Show.Name") != "Some Show Name" # the trap it avoids +# ── §10.1/V14: telling a real episode marker from a mangled number ────────── + +def test_has_episode_marker_accepts_real_markers(): + for name in ["Some.Show.S01E08.mkv", "some.show.s1.e8.mkv", + "Some Show 1x08.mkv", "Some Show 01x08.mkv", + "Some Show Episode 8.mkv", "Some Show ep08.mkv", + "Some Show ep.8.mkv", "Some Show Season 1.mkv", + "Une Serie Saison 3.mkv"]: + assert has_episode_marker(name), name + + +def test_has_episode_marker_rejects_bare_numbers_and_ordinary_words(): + # "1080p" truncated to "108", "1280" left in, a year, plain words that + # merely contain "ep" — none of these are episode markers. + for name in ["Some.Film.2017.MULTI.108.grp.mkv", + "Some.Flick.2013.1280.x264-grp.mkv", + "Some Movie 2017.mkv", "The Dark Knight.mkv", + "Sleep 8.mkv", "Deep 8 mm.mkv", "Ocean's 11.mkv"]: + assert not has_episode_marker(name), name + + # ── bug 2026-08-29: guessit peels "Volume N" off the title ───────────────── def test_movie_volume_number_is_folded_back_into_the_title(): -- cgit v1.2.3