diff options
| author | Christophe Besson <cbesson@gmail.com> | 2026-08-26 17:18:37 +0200 |
|---|---|---|
| committer | Christophe Besson <cbesson@gmail.com> | 2026-08-26 17:18:37 +0200 |
| commit | 80cb6c4dc5f22336387f2ac74ef2cb85cf2cf7d4 (patch) | |
| tree | 590e2daf81b2feef674f1bd580b9bf40f83addad /packages/meshbay-node/src/meshbay_node/indexer | |
| parent | ca62b4123b854fc9ece258e14c84c895aaa275cd (diff) | |
| parent | 59289b82dc8af08c605fd247c90a5c5ec92c6db3 (diff) | |
| download | meshbay-80cb6c4dc5f22336387f2ac74ef2cb85cf2cf7d4.tar.gz | |
Merge branch 'feat/video-type-filter'
Videos app grouping and TMDB matching fixes, plus a new toolbar filter:
- A season-like ancestor folder (numbered season, or Specials/Bonus/Extras
-> season 0) now names the show from its own root folder, unconditionally
— never a per-file guessit title, which cannot tell a show's real name
from an individual episode's own one-off name when the filename carries
no reliable ShowName/SxxExx structure. Fixes a real show's episodes and
Specials folder alike showing up as dozens of individual "movies", each
matched against TMDB by its own one-off title.
- The ancestor walk continues past every consecutive season-like folder,
not just the first — a per-season Bonus folder is nested two levels
inside the show, and stopping at the first would name the season as the
show.
- A season spanning more than one folder (a per-book Bonus folder nested
inside every numbered season) no longer hands out colliding episode
numbers independently in each one.
- guessit's own episode number is not trusted when it comes from a bare
3-digit leading number ("100" parses as season=1/episode=0, not
episode=100) — read directly via regex instead.
- Recognizes "S1"/"S2"-style abbreviated season folders, not just full
words ("Season"/"Saison"/"Livre") — a real show organized its later
seasons this way and they never got the ancestor-based grouping fix at
all.
- The TMDB match cache is no longer trusted across a movie<->show
reclassification it doesn't know happened.
- New: an All/Movies/Series filter in the Videos toolbar, to the left of
the search field, defaulting to "All"; wraps correctly on mobile.
Verified live against the real libraries these were found on throughout —
not synthetic reproduction alone. 430 hub tests + 667 node tests passing.
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/indexer')
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/indexer/enrich.py | 143 | ||||
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/indexer/title_parse.py | 32 |
2 files changed, 160 insertions, 15 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index ae3f3dc..07f0be7 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -36,17 +36,52 @@ MAX_ANCESTOR_DEPTH = 4 # Bounds the "borrow a title from a sibling episode filename" scan (§3.4) so # a folder with thousands of files costs a fixed, small amount of work. MAX_SIBLINGS_CHECKED = 20 +# Bounds the whole-show scan _synthetic_episode_number uses to rank a +# season's files across more than one folder (§3.4c) — a show's total file +# count, not just one folder's, so this needs more headroom than +# MAX_SIBLINGS_CHECKED. +MAX_SEASON_FILES_CHECKED = 500 -def _season_from_ancestors(file_path: Path) -> int | None: +def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, Path] | None: + """ + §3.4/§3.4c: walk up ancestor folders for a season-like one (a numbered + season, or Specials/Bonus/Extras -> season 0) — season from the first + (innermost) match, but the show's own name from *above every + consecutive season-like ancestor*, not just the first one. A + per-season Bonus folder (`Show/Season N/Bonus/file.ext`) is nested two + levels inside the show, both of them season-like on their own + ("Bonus" and "Season N") — stopping at the first would hand back + "Season N" as the show's name instead of "Show". + + Trusted over any per-file guessit title once found: a bare episode + numbering convention with no show name embedded at all + (`001 Episode's Own Title.mkv`, no SxxExx, no show prefix) is + completely ordinary and makes guessit invent a title from whatever + text follows the number — that text is the individual episode's own + name, never the show's, and every episode in the folder produces a + different one. None of that ambiguity exists for the folder structure + itself: the show's own root folder names it once, not per file. + + None if no ancestor looks like a season folder at all — a flat + library has nothing to borrow a show name from this way, and + _title_from_siblings' per-file logic is what applies instead. + """ folder = file_path.parent - for _ in range(MAX_ANCESTOR_DEPTH): - if folder is None or folder == folder.parent: - break - season = title_parse.season_from_folder_name(folder.name) - if season is not None: - return season + season: int | None = None + depth = 0 + while folder is not None and folder.parent != folder and depth < MAX_ANCESTOR_DEPTH: + this_season = title_parse.season_from_folder_name(folder.name) + if this_season is None: + if season is None: + folder = folder.parent + depth += 1 + continue + return season, folder + if season is None: + season = this_season folder = folder.parent + depth += 1 return None @@ -55,6 +90,13 @@ def _title_from_siblings(file_path: Path) -> str | None: §3.4: an episode filename with no show name in it borrows the title from a representative sibling in the same folder, never from the folder name alone (an acronym-named show folder is a real, observed case). + + Requires the sibling to carry its own episode number too, not just a + title — a folder where every file is a one-off-named Special (§3.4b) + has plenty of `display_title`s (guessit reads *a* title off nearly + anything) but none of them name the show; requiring a real episode + number alongside is what tells apart a genuinely representative sibling + from another Special just like this one. """ try: names = sorted(p.name for p in file_path.parent.iterdir() if p.is_file()) @@ -70,11 +112,52 @@ def _title_from_siblings(file_path: Path) -> str | None: if checked > MAX_SIBLINGS_CHECKED: break parsed = title_parse.parse_episode_filename(name) - if parsed.display_title: + if parsed.display_title and parsed.episode is not None: return parsed.display_title return None +def _synthetic_episode_number(file_path: Path, show_root: Path, season: int) -> int: + """ + §3.4b/§3.4c: a Specials/Bonus folder's files often carry no episode + number at all — each is just named after its own one-off title. The + frontend (video-app.js's buildSeasons) sorts within a season by this + number but only needs it to provide a stable order, not to mean + anything beyond that. + + Ranked across the *whole show*, not just this file's own folder: season + 0 routinely spans more than one folder under the show's root — a + Bonus folder nested inside every numbered season + (`Show/Season N/Bonus/...`) is exactly the shape that produces this — + and ranking within just this file's own folder hands out the same + "episode 1" again in every other one, found live as three unrelated + Specials all showing up as "S0E01". Deterministic across re-scans as + long as the show's contents don't change. 1-based so it reads as + "episode 1", not "episode 0", in a UI that already uses season 0 for + "Specials" itself. Bounded (MAX_SEASON_FILES_CHECKED) so a huge show + costs a fixed amount of work rather than scaling with its size. + """ + try: + candidates = sorted( + (p for p in show_root.rglob("*") + if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]), + key=lambda p: (str(p.parent), p.name), + ) + except OSError: + return 1 + rank = 0 + for i, candidate in enumerate(candidates): + if i >= MAX_SEASON_FILES_CHECKED: + break + ancestor = _season_and_show_from_ancestors(candidate) + if ancestor is None or ancestor[0] != season: + continue + rank += 1 + if candidate == file_path: + return rank + return 1 + + async def _make_thumbnail(file_path: Path, duration: float | None) -> bytes | None: """One ffmpeg frame grab at ~10% of duration (or 5s if unknown), scaled down.""" seek = max(0.0, (duration or 50.0) * 0.1) @@ -165,14 +248,48 @@ class Enricher: entry.id, fields.get("duration"), fields.get("width"), fields.get("height")) ep = title_parse.parse_episode_filename(entry.name) - if ep.episode is not None: + ancestor = await asyncio.to_thread(_season_and_show_from_ancestors, file_path) + + if ancestor is not None: + # A season-like ancestor folder exists — the show's own + # root folder names it, trusted over any per-file guessit + # title. This is what actually groups every episode under + # one show: a bare numbering convention with no show name + # in the filename at all (`001 Episode's Own Title.mkv`, + # ordinary enough on its own) makes guessit invent a title + # from whatever text follows the number, which is that + # episode's own name, never the show's — and differs for + # every episode, so nothing would ever group together. + # Found live: a real show's episodes and its Specials + # folder alike, both named this way, showed up as + # individual "movies" each searched against — and matched + # to — an unrelated real film sharing that one-off title, + # instead of anything grouped under the show at all. + season, show_folder = ancestor + # Not ep.episode: guessit reads a bare 3-digit leading + # number (routine once a season-like ancestor is already + # doing the real season/episode work) as a concatenated + # season+episode guess rather than a plain episode number + # — confirmed live, "100 Title.mkv" parsed as episode 0, + # not 100, with nothing in guessit's own output telling + # that apart from a real 2-digit episode. The direct regex + # reads the whole leading number as it is. + episode = title_parse.leading_episode_number(entry.name) + if episode is None: + episode = ep.episode + if episode is None: + episode = await asyncio.to_thread( + _synthetic_episode_number, file_path, show_folder, season) + fields["display_title"] = show_folder.name + fields["season"] = season + fields["episode"] = episode + elif ep.episode is not None: + # No season-like ancestor at all (a flat library) but the + # filename itself carries season+episode (§3.4). title = ep.display_title or await asyncio.to_thread( _title_from_siblings, file_path) - season = ep.season - if season is None: - season = await asyncio.to_thread(_season_from_ancestors, file_path) fields["display_title"] = title or title_parse.naive_title(entry.name) - fields["season"] = season + fields["season"] = ep.season fields["episode"] = ep.episode else: mv = title_parse.parse_movie_filename(entry.name) diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 5203247..24b423e 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -36,13 +36,21 @@ _EDITION_RE = re.compile("|".join(_EDITION_PHRASES), re.IGNORECASE) # A season-like ancestor folder: the English/French words plus a number or # Roman numeral. Vocabulary is a plain tuple so a deployment can extend it -# per locale without touching the regex-building logic. -SEASON_WORDS = ("season", "saison") +# per locale without touching the regex-building logic. "livre" ("book") is +# real, observed vocabulary too — some shows name their seasons that way +# (Roman numerals: "Livre I".."Livre VI") rather than "saison". +SEASON_WORDS = ("season", "saison", "livre") _SEASON_RE = re.compile( r"(?:" + "|".join(SEASON_WORDS) + r")\s*([0-9]+|[ivxlc]+)\b", re.IGNORECASE, ) _SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE) +# A bare "S" + number as the *whole* folder name — "S1", "S2", "S02" — a +# common abbreviated convention distinct from SEASON_WORDS' full words. +# Anchored to the entire name, not just `\b`-bounded within a longer +# string, so it only matches a folder actually named just that — never +# some other word that merely starts with "s" followed by digits. +_SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE) _ROMAN_NUMERALS = { 2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI", @@ -108,6 +116,9 @@ def season_from_folder_name(name: str) -> int | None: """ if _SPECIALS_RE.search(name): return 0 + m = _SEASON_ABBREV_RE.match(name.strip()) + if m: + return int(m.group(1)) m = _SEASON_RE.search(name) if not m: return None @@ -227,3 +238,20 @@ def parse_episode_filename(filename: str) -> ParsedName: display_title=title or None, naive_title=nt, season=season, episode=episode, confidence=confidence, ) + + +# A bare leading episode number, no show name attached (§3.4c) — the same +# shape as music's _TRACK_PREFIX_RE, capped at 3 digits for the same reason: +# a leading year ("2010 - Episode.mkv") is 4 digits and must not match. +# guessit's own `episode` is not a substitute here: given exactly 3 digits it +# tries to read them as a concatenated SxxE/SEE season+episode pair instead +# of a plain episode number — confirmed live, "100 Title.mkv" parses as +# season=1, episode=0, not episode=100 — silently wrong in a way nothing +# about its output distinguishes from a real 2-digit episode. This reads the +# whole leading number as one value instead. +_LEADING_NUMBER_RE = re.compile(r"^(\d{1,3})[\s._-]+(?=\S)") + + +def leading_episode_number(filename: str) -> int | None: + m = _LEADING_NUMBER_RE.match(filename) + return int(m.group(1)) if m else None |