diff options
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node')
3 files changed, 64 insertions, 13 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py index 07f0be7..8b2eca5 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/enrich.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/enrich.py @@ -283,9 +283,18 @@ class Enricher: fields["display_title"] = show_folder.name fields["season"] = season fields["episode"] = episode - elif ep.episode is not None: + elif (ep.season is not None and ep.episode is not None + and (title_parse.has_episode_marker(entry.name) + or title_parse.year_in(entry.name) is None)): # No season-like ancestor at all (a flat library) but the - # filename itself carries season+episode (§3.4). + # filename itself carries season+episode (§3.4) — *and* it + # is a real marker, not guessit reading a bare number as + # SxxExx. A movie whose "1080p" tag was truncated to "108", + # or "1280" left in the name, otherwise parses to S01E08 / + # S12E80 and gets shelved as a nonexistent series + # (found live 2026-08-29). A genuine flat-dumped episode + # has an explicit SxxExx/1x08/"Episode N" marker; a movie + # has a "(2019)"-style year and no such marker. title = ep.display_title or await asyncio.to_thread( _title_from_siblings, file_path) fields["display_title"] = title or title_parse.naive_title(entry.name) diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 018746b..def947f 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -288,6 +288,26 @@ def strip_track_prefix(text: str) -> str: return re.sub(r"^[\s-]+", "", text[m.end():]).strip() if m else text +# An *explicit* season/episode marker: SxxExx, 1x08, "Episode 8", "Ep 8", +# "Season 1"/"Saison 1". guessit will also invent a season+episode from a +# bare 3-4 digit run ("1080p" truncated to "108" -> S01E08; "1280" -> +# S12E80), which is how a plain movie ends up shelved as a series +# (§10.1/V14). The indexer uses this to tell a real flat-library episode +# from that hallucination. +_EPISODE_MARKER_RE = re.compile( + r"s\d{1,2}[\s._-]*e\d{1,3}" + r"|\b\d{1,2}x\d{1,3}\b" + r"|\bepisode[\s._-]*\d{1,3}\b" + r"|\bep[\s._-]*\d{1,3}\b" + r"|\b(?:season|saison)[\s._-]*\d{1,2}\b", + re.IGNORECASE, +) + + +def has_episode_marker(filename: str) -> bool: + return bool(_EPISODE_MARKER_RE.search(filename)) + + def parse_episode_filename(filename: str) -> ParsedName: """ Parse an episode filename. `display_title` may come back None (e.g. diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py index af6bf08..1bd7203 100644 --- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py +++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py @@ -2915,6 +2915,17 @@ class WebRTCPeerSession: "file_id": file_id, "confidence": 0}) return + # A video the indexer has seen but not yet *enriched* has no + # display_title (enrich.py always sets one) and season/episode still + # None — so the movie/show split reads "movie" and would hand its raw + # filename to TMDB's movie search. During a slow initial scan with a + # browser on the Videos tab that is a storm of + # `search/movie?query=<raw filename>` (found live 2026-08-29, an + # 8-minute scan). While un-enriched we never *search*: we serve a + # cached match if there is one (§ below), else confidence 0 and the + # client refetches once the index delta carries the enriched fields. + enriched = bool(entry.display_title) + is_show = entry.season is not None and entry.episode is not None media_type = "tv" if is_show else "movie" @@ -2923,21 +2934,32 @@ class WebRTCPeerSession: tmdb_id = None if cached is not None: cached_tmdb_id, cached_media_type = cached - # Trustworthy only if it still agrees with what this file - # resolves to *now*. season/episode come from index-time - # enrichment (enrich.py), which can reclassify a file between - # movie and show on a later scan without this cache knowing — - # it is keyed by the file's content hash alone, which a - # reclassification never changes. Found live: an enrichment fix - # to a Specials-folder bug reclassified hundreds of files from - # "movie" to "tv", and every one kept answering with its - # stale movie-era match forever, because this was trusted - # before ever comparing media_type against the current one. - if cached_media_type == media_type: + # Serve the cached match when its kind still agrees with the + # entry's current classification — OR when the entry is not + # enriched yet: its season/episode aren't populated, so the + # movie/show split above is not meaningful, and the cached kind + # (set when this file WAS enriched) is the reliable one. This is + # what keeps a restart from re-querying TMDB for everything + # already resolved: the storm was an un-enriched show episode + # looking like a "movie" and treating its own valid "tv" match + # as stale. + # + # Once enriched, the strict `cached_media_type == media_type` + # check still stands: an enrichment fix that reclassifies a + # folder movie->tv must drop the stale movie-era match and + # re-resolve (found live — a Specials-folder fix left hundreds + # of files answering with their wrong-kind match forever). + if cached_media_type == media_type or not enriched: + media_type = cached_media_type + is_show = media_type == "tv" tmdb_id = cached_tmdb_id meta = await media_cache.get_tmdb_meta(tmdb_id, media_type) if meta is None: + if not enriched: + self._send({"type": MNP.MEDIA_META_RESP, "v": MNP_VERSION, + "file_id": file_id, "confidence": 0}) + return result, ratio = await self._tmdb_search(tmdb_client, entry, is_show) if result is None or ratio < 0.6: self._send({"type": MNP.MEDIA_META_RESP, "v": MNP_VERSION, |