aboutsummaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node/transport
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-08-29 15:43:38 +0200
committerChristophe Besson <cbesson@gmail.com>2026-08-29 15:43:38 +0200
commit5d28d0c96cc8489645178b483835489779cb3887 (patch)
tree90248f1c3211ca37495a774d0bc140109c299abc /packages/meshbay-node/src/meshbay_node/transport
parent236e5e811355212945b89c4f7a99df5c837e97c7 (diff)
downloadmeshbay-5d28d0c96cc8489645178b483835489779cb3887.tar.gz
feat(node): V8–V11 — show-branch ladder, year-aware _best_match, wider sequel_variants
V8: the TV/show branch of _tmdb_search used the old "first candidate over 0.6 wins" shape. It now shares one _tmdb_ladder helper with the movie branch — score every candidate query, keep the best, fast-path a confident primary hit. A year lifted off the show's folder name (title_parse.year_in, e.g. "Some.Show.2022.S01") rescues a sub-0.6 hit that lands on the exact year. title_parse.clean_query de-dots a folder-derived title without naive_title's extension-stripping trap. V9: _best_match gains an optional `year`. When the top result is not a confident textual hit (ratio < 0.6) and a year was requested, a different result of that exact release year is preferred — TMDB already year-filtered the search, so this is a hard corroboration, not the fuzzy re-rank §3.3 warns against. A confident top hit is never overridden. search_movie/search_tv forward the year. V10: sequel_variants widened — trailing Roman→digit as well as digit→Roman, spelled-out indices (one..twelve / un..douze / ordinals), and a "Part N" / "Chapitre N" wrapper. Still empty for a trailing word that is not an index or a 4-digit year. V11: when the primary hit is already decent (>= 0.6) and there is nothing more specific to try (no alternative_title, no sequel variant — only a punctuation restatement left), the ladder returns without the extra requests. The clean-title common case is back to one call. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018BMLQjqFGCize2KtNBT79v
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/transport')
-rw-r--r--packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py116
1 files changed, 62 insertions, 54 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
index 788a8f1..c75819e 100644
--- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
+++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py
@@ -3173,80 +3173,88 @@ class WebRTCPeerSession:
async def _tmdb_search(self, tmdb_client, entry, is_show: bool):
"""
- §3.3's retry ladder. TMDB's own top result is still trusted per
- query (§3.3's last row — no local re-ranking of *its* list); what
- changed is that the ladder now *scores every candidate query* and
- keeps the best, instead of returning the first that merely clears
- 0.6.
+ §3.3's retry ladder — same shape for movies and shows (§10.1/V8).
+ TMDB's own top result is still trusted per query (§3.3's last row —
+ no local re-ranking of *its* list); what the ladder adds is that it
+ *scores every candidate query* and keeps the best, instead of
+ returning the first that merely clears 0.6.
The bare parsed title is the weakest query: guessit drops a
- "Volume 2", strips a real subtitle into `alternative_title`, and
- renders a sequel number where TMDB uses a Roman numeral. A wrong
- film that happened to score ~0.7 against that weak query — a
- same-year making-of documentary, or a franchise entry whose
- localized TMDB title *is* the franchise name — used to win outright
- before `alternative_title` or the Roman-numeral variant was ever
- tried. Found live (2026-08-29): a numbered sequel matched a
- same-year documentary; a two-volume film's second part matched the
- first; several franchise entries matched one early entry.
+ "Volume 2", strips a real subtitle into `alternative_title`, renders
+ a sequel index where TMDB spells it differently, and a show's folder
+ name can carry a year or release-group noise. A wrong entry that
+ scored ~0.7 against that weak query — a same-year making-of
+ documentary, a franchise entry whose localized TMDB title *is* the
+ franchise name, a season-specific promo entry standing in for a
+ whole show — used to win outright before a stronger candidate was
+ ever tried. Found live (2026-08-29).
"""
from meshbay_node.indexer import title_parse
if is_show:
title = entry.display_title or title_parse.naive_title(entry.name)
- result, ratio = await tmdb_client.search_tv(title)
- if result is None or ratio < 0.6:
- naive = title_parse.naive_title(entry.name)
- if naive != title:
- result, ratio = await tmdb_client.search_tv(naive)
- return result, ratio
-
- def _release_year(res: dict) -> int | None:
- d = str(res.get("release_date") or res.get("first_air_date") or "")
- return int(d[:4]) if d[:4].isdigit() else None
+ name_naive = title_parse.naive_title(entry.name)
+ year = title_parse.year_in(title) or title_parse.year_in(entry.name)
+ extra = [c for c in (name_naive, title_parse.clean_query(title))
+ if c and c != title]
+ return await self._tmdb_ladder(
+ tmdb_client.search_tv, title, extra, year, strong_extra=False)
parsed = title_parse.parse_movie_filename(entry.name)
title = entry.display_title or parsed.display_title or parsed.naive_title
+ strong = [c for c in (parsed.alt_title, *title_parse.sequel_variants(title)) if c]
+ extra = [c for c in (*strong, parsed.naive_title) if c and c != title]
+ return await self._tmdb_ladder(
+ tmdb_client.search_movie, title, extra, parsed.year,
+ strong_extra=bool(strong))
- # Fast path, unchanged in effect: a strong direct hit still returns
- # on the first call, so the common case costs exactly one request
- # and the new ladder below only engages in the ambiguous 0<ratio<0.85
- # zone where every one of the live bugs lived.
- result, ratio = await tmdb_client.search_movie(title, parsed.year)
- if result is not None and ratio >= 0.85:
- return result, ratio
-
- best_result, best_score = (result, ratio) if result is not None else (None, 0.0)
+ @staticmethod
+ async def _tmdb_ladder(search_fn, primary: str, extra: list[str],
+ year: int | None, strong_extra: bool):
+ """
+ `search_fn(query, year) -> (result|None, ratio)`. Try `primary`,
+ return at once on a confident hit (ratio >= 0.85 — the common case,
+ one request). Otherwise score each `extra` candidate and keep the
+ best. `strong_extra` says whether `extra` contains anything more
+ specific than a punctuation-normalised restatement of `primary`
+ (an alternative_title, a sequel variant); when it does not and the
+ primary hit is already decent, the remaining calls are skipped
+ (§10.1/V11 — they almost never win and cost a round trip each).
+ """
+ def _year_of(res: dict) -> int | None:
+ d = str(res.get("release_date") or res.get("first_air_date") or "")
+ return int(d[:4]) if d[:4].isdigit() else None
- def _apply_year_rescue(res: dict, r: float) -> float:
- # Only for a query whose own top hit is weak on its face
- # (ratio < 0.6): TMDB already year-filtered the search, so its
- # top result landing exactly on the filename's year is a hard
- # corroborating signal that the low string ratio is a
- # localized/rearranged title, not a wrong film. Never lets year
- # equality outrank a genuinely strong textual match elsewhere.
- if r < 0.6 and parsed.year and _release_year(res) == parsed.year:
+ def _rescue(res: dict, r: float) -> float:
+ # A sub-0.6 hit whose result lands on the exact requested year:
+ # TMDB already year-filtered the search, so this is a hard
+ # corroboration that the low ratio is a localised/rearranged
+ # title, not a wrong entry. Never overrides a confident hit.
+ if r < 0.6 and year and _year_of(res) == year:
return max(r, 0.6)
return r
+ result, ratio = await search_fn(primary, year)
+ if result is not None and ratio >= 0.85:
+ return result, ratio
+ best_result, best_score = (result, ratio) if result is not None else (None, 0.0)
if best_result is not None:
- best_score = _apply_year_rescue(best_result, best_score)
+ best_score = _rescue(best_result, best_score)
+ if best_score >= 0.6 and not strong_extra:
+ return best_result, best_score
- for candidate in filter(None, [parsed.alt_title,
- *title_parse.sequel_variants(title),
- parsed.naive_title]):
- if candidate == title:
+ for candidate in extra:
+ if candidate == primary:
continue
- r2, ratio2 = await tmdb_client.search_movie(candidate, parsed.year)
- if r2 is None and parsed.year:
- # A year-filtered search that finds nothing: the filename's
- # year tag may be an edition/regional year TMDB doesn't
- # carry. Retry the same candidate unconstrained before
- # dropping it.
- r2, ratio2 = await tmdb_client.search_movie(candidate)
+ r2, ratio2 = await search_fn(candidate, year)
+ if r2 is None and year:
+ # A year-filtered search that finds nothing: the year tag
+ # may be an edition/regional year TMDB doesn't carry. Retry
+ # the candidate unconstrained before dropping it.
+ r2, ratio2 = await search_fn(candidate, None)
if r2 is None:
continue
- score2 = _apply_year_rescue(r2, ratio2)
+ score2 = _rescue(r2, ratio2)
if score2 > best_score:
best_result, best_score = r2, score2
if best_score >= 0.85: