summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src
diff options
context:
space:
mode:
Diffstat (limited to 'packages/meshbay-node/src')
-rw-r--r--packages/meshbay-node/src/meshbay_node/indexer/enrich.py143
-rw-r--r--packages/meshbay-node/src/meshbay_node/indexer/title_parse.py32
-rw-r--r--packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py16
3 files changed, 174 insertions, 17 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py
index ae3f3dc..07f0be7 100644
--- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py
+++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py
@@ -36,17 +36,52 @@ MAX_ANCESTOR_DEPTH = 4
# Bounds the "borrow a title from a sibling episode filename" scan (§3.4) so
# a folder with thousands of files costs a fixed, small amount of work.
MAX_SIBLINGS_CHECKED = 20
+# Bounds the whole-show scan _synthetic_episode_number uses to rank a
+# season's files across more than one folder (§3.4c) — a show's total file
+# count, not just one folder's, so this needs more headroom than
+# MAX_SIBLINGS_CHECKED.
+MAX_SEASON_FILES_CHECKED = 500
-def _season_from_ancestors(file_path: Path) -> int | None:
+def _season_and_show_from_ancestors(file_path: Path) -> tuple[int, Path] | None:
+ """
+ §3.4/§3.4c: walk up ancestor folders for a season-like one (a numbered
+ season, or Specials/Bonus/Extras -> season 0) — season from the first
+ (innermost) match, but the show's own name from *above every
+ consecutive season-like ancestor*, not just the first one. A
+ per-season Bonus folder (`Show/Season N/Bonus/file.ext`) is nested two
+ levels inside the show, both of them season-like on their own
+ ("Bonus" and "Season N") — stopping at the first would hand back
+ "Season N" as the show's name instead of "Show".
+
+ Trusted over any per-file guessit title once found: a bare episode
+ numbering convention with no show name embedded at all
+ (`001 Episode's Own Title.mkv`, no SxxExx, no show prefix) is
+ completely ordinary and makes guessit invent a title from whatever
+ text follows the number — that text is the individual episode's own
+ name, never the show's, and every episode in the folder produces a
+ different one. None of that ambiguity exists for the folder structure
+ itself: the show's own root folder names it once, not per file.
+
+ None if no ancestor looks like a season folder at all — a flat
+ library has nothing to borrow a show name from this way, and
+ _title_from_siblings' per-file logic is what applies instead.
+ """
folder = file_path.parent
- for _ in range(MAX_ANCESTOR_DEPTH):
- if folder is None or folder == folder.parent:
- break
- season = title_parse.season_from_folder_name(folder.name)
- if season is not None:
- return season
+ season: int | None = None
+ depth = 0
+ while folder is not None and folder.parent != folder and depth < MAX_ANCESTOR_DEPTH:
+ this_season = title_parse.season_from_folder_name(folder.name)
+ if this_season is None:
+ if season is None:
+ folder = folder.parent
+ depth += 1
+ continue
+ return season, folder
+ if season is None:
+ season = this_season
folder = folder.parent
+ depth += 1
return None
@@ -55,6 +90,13 @@ def _title_from_siblings(file_path: Path) -> str | None:
§3.4: an episode filename with no show name in it borrows the title from
a representative sibling in the same folder, never from the folder name
alone (an acronym-named show folder is a real, observed case).
+
+ Requires the sibling to carry its own episode number too, not just a
+ title — a folder where every file is a one-off-named Special (§3.4b)
+ has plenty of `display_title`s (guessit reads *a* title off nearly
+ anything) but none of them name the show; requiring a real episode
+ number alongside is what tells apart a genuinely representative sibling
+ from another Special just like this one.
"""
try:
names = sorted(p.name for p in file_path.parent.iterdir() if p.is_file())
@@ -70,11 +112,52 @@ def _title_from_siblings(file_path: Path) -> str | None:
if checked > MAX_SIBLINGS_CHECKED:
break
parsed = title_parse.parse_episode_filename(name)
- if parsed.display_title:
+ if parsed.display_title and parsed.episode is not None:
return parsed.display_title
return None
+def _synthetic_episode_number(file_path: Path, show_root: Path, season: int) -> int:
+ """
+ §3.4b/§3.4c: a Specials/Bonus folder's files often carry no episode
+ number at all — each is just named after its own one-off title. The
+ frontend (video-app.js's buildSeasons) sorts within a season by this
+ number but only needs it to provide a stable order, not to mean
+ anything beyond that.
+
+ Ranked across the *whole show*, not just this file's own folder: season
+ 0 routinely spans more than one folder under the show's root — a
+ Bonus folder nested inside every numbered season
+ (`Show/Season N/Bonus/...`) is exactly the shape that produces this —
+ and ranking within just this file's own folder hands out the same
+ "episode 1" again in every other one, found live as three unrelated
+ Specials all showing up as "S0E01". Deterministic across re-scans as
+ long as the show's contents don't change. 1-based so it reads as
+ "episode 1", not "episode 0", in a UI that already uses season 0 for
+ "Specials" itself. Bounded (MAX_SEASON_FILES_CHECKED) so a huge show
+ costs a fixed amount of work rather than scaling with its size.
+ """
+ try:
+ candidates = sorted(
+ (p for p in show_root.rglob("*")
+ if p.is_file() and p.suffix.lower() in MEDIA_EXTENSIONS["video"]),
+ key=lambda p: (str(p.parent), p.name),
+ )
+ except OSError:
+ return 1
+ rank = 0
+ for i, candidate in enumerate(candidates):
+ if i >= MAX_SEASON_FILES_CHECKED:
+ break
+ ancestor = _season_and_show_from_ancestors(candidate)
+ if ancestor is None or ancestor[0] != season:
+ continue
+ rank += 1
+ if candidate == file_path:
+ return rank
+ return 1
+
+
async def _make_thumbnail(file_path: Path, duration: float | None) -> bytes | None:
"""One ffmpeg frame grab at ~10% of duration (or 5s if unknown), scaled down."""
seek = max(0.0, (duration or 50.0) * 0.1)
@@ -165,14 +248,48 @@ class Enricher:
entry.id, fields.get("duration"), fields.get("width"), fields.get("height"))
ep = title_parse.parse_episode_filename(entry.name)
- if ep.episode is not None:
+ ancestor = await asyncio.to_thread(_season_and_show_from_ancestors, file_path)
+
+ if ancestor is not None:
+ # A season-like ancestor folder exists — the show's own
+ # root folder names it, trusted over any per-file guessit
+ # title. This is what actually groups every episode under
+ # one show: a bare numbering convention with no show name
+ # in the filename at all (`001 Episode's Own Title.mkv`,
+ # ordinary enough on its own) makes guessit invent a title
+ # from whatever text follows the number, which is that
+ # episode's own name, never the show's — and differs for
+ # every episode, so nothing would ever group together.
+ # Found live: a real show's episodes and its Specials
+ # folder alike, both named this way, showed up as
+ # individual "movies" each searched against — and matched
+ # to — an unrelated real film sharing that one-off title,
+ # instead of anything grouped under the show at all.
+ season, show_folder = ancestor
+ # Not ep.episode: guessit reads a bare 3-digit leading
+ # number (routine once a season-like ancestor is already
+ # doing the real season/episode work) as a concatenated
+ # season+episode guess rather than a plain episode number
+ # — confirmed live, "100 Title.mkv" parsed as episode 0,
+ # not 100, with nothing in guessit's own output telling
+ # that apart from a real 2-digit episode. The direct regex
+ # reads the whole leading number as it is.
+ episode = title_parse.leading_episode_number(entry.name)
+ if episode is None:
+ episode = ep.episode
+ if episode is None:
+ episode = await asyncio.to_thread(
+ _synthetic_episode_number, file_path, show_folder, season)
+ fields["display_title"] = show_folder.name
+ fields["season"] = season
+ fields["episode"] = episode
+ elif ep.episode is not None:
+ # No season-like ancestor at all (a flat library) but the
+ # filename itself carries season+episode (§3.4).
title = ep.display_title or await asyncio.to_thread(
_title_from_siblings, file_path)
- season = ep.season
- if season is None:
- season = await asyncio.to_thread(_season_from_ancestors, file_path)
fields["display_title"] = title or title_parse.naive_title(entry.name)
- fields["season"] = season
+ fields["season"] = ep.season
fields["episode"] = ep.episode
else:
mv = title_parse.parse_movie_filename(entry.name)
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py
index 5203247..24b423e 100644
--- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py
+++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py
@@ -36,13 +36,21 @@ _EDITION_RE = re.compile("|".join(_EDITION_PHRASES), re.IGNORECASE)
# A season-like ancestor folder: the English/French words plus a number or
# Roman numeral. Vocabulary is a plain tuple so a deployment can extend it
-# per locale without touching the regex-building logic.
-SEASON_WORDS = ("season", "saison")
+# per locale without touching the regex-building logic. "livre" ("book") is
+# real, observed vocabulary too — some shows name their seasons that way
+# (Roman numerals: "Livre I".."Livre VI") rather than "saison".
+SEASON_WORDS = ("season", "saison", "livre")
_SEASON_RE = re.compile(
r"(?:" + "|".join(SEASON_WORDS) + r")\s*([0-9]+|[ivxlc]+)\b",
re.IGNORECASE,
)
_SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE)
+# A bare "S" + number as the *whole* folder name — "S1", "S2", "S02" — a
+# common abbreviated convention distinct from SEASON_WORDS' full words.
+# Anchored to the entire name, not just `\b`-bounded within a longer
+# string, so it only matches a folder actually named just that — never
+# some other word that merely starts with "s" followed by digits.
+_SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE)
_ROMAN_NUMERALS = {
2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI",
@@ -108,6 +116,9 @@ def season_from_folder_name(name: str) -> int | None:
"""
if _SPECIALS_RE.search(name):
return 0
+ m = _SEASON_ABBREV_RE.match(name.strip())
+ if m:
+ return int(m.group(1))
m = _SEASON_RE.search(name)
if not m:
return None
@@ -227,3 +238,20 @@ def parse_episode_filename(filename: str) -> ParsedName:
display_title=title or None, naive_title=nt,
season=season, episode=episode, confidence=confidence,
)
+
+
+# A bare leading episode number, no show name attached (§3.4c) — the same
+# shape as music's _TRACK_PREFIX_RE, capped at 3 digits for the same reason:
+# a leading year ("2010 - Episode.mkv") is 4 digits and must not match.
+# guessit's own `episode` is not a substitute here: given exactly 3 digits it
+# tries to read them as a concatenated SxxE/SEE season+episode pair instead
+# of a plain episode number — confirmed live, "100 Title.mkv" parses as
+# season=1, episode=0, not episode=100 — silently wrong in a way nothing
+# about its output distinguishes from a real 2-digit episode. This reads the
+# whole leading number as one value instead.
+_LEADING_NUMBER_RE = re.compile(r"^(\d{1,3})[\s._-]+(?=\S)")
+
+
+def leading_episode_number(filename: str) -> int | None:
+ m = _LEADING_NUMBER_RE.match(filename)
+ return int(m.group(1)) if m else None
diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
index f95e59e..252e202 100644
--- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
+++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
@@ -2914,8 +2914,20 @@ class WebRTCPeerSession:
meta = None
tmdb_id = None
if cached is not None:
- tmdb_id, media_type = cached
- meta = await media_cache.get_tmdb_meta(tmdb_id, media_type)
+ cached_tmdb_id, cached_media_type = cached
+ # Trustworthy only if it still agrees with what this file
+ # resolves to *now*. season/episode come from index-time
+ # enrichment (enrich.py), which can reclassify a file between
+ # movie and show on a later scan without this cache knowing —
+ # it is keyed by the file's content hash alone, which a
+ # reclassification never changes. Found live: an enrichment fix
+ # to a Specials-folder bug reclassified hundreds of files from
+ # "movie" to "tv", and every one kept answering with its
+ # stale movie-era match forever, because this was trusted
+ # before ever comparing media_type against the current one.
+ if cached_media_type == media_type:
+ tmdb_id = cached_tmdb_id
+ meta = await media_cache.get_tmdb_meta(tmdb_id, media_type)
if meta is None:
result, ratio = await self._tmdb_search(tmdb_client, entry, is_show)