diff options
Diffstat (limited to 'packages/meshbay-node')
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/indexer/title_parse.py | 100 | ||||
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/media_cache.py | 12 | ||||
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/tmdb.py | 58 | ||||
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py | 165 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_media_cache.py | 13 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_title_parse.py | 44 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_tmdb.py | 53 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_tmdb_rematch_policy.py | 147 | ||||
| -rw-r--r-- | packages/meshbay-node/tests/test_tmdb_search_ladder.py | 58 |
9 files changed, 561 insertions, 89 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 24c23bc..018746b 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -52,14 +52,8 @@ _SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE) # some other word that merely starts with "s" followed by digits. _SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE) -_ROMAN_NUMERALS = { - 2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI", - 7: "VII", 8: "VIII", 9: "IX", 10: "X", -} - - def _roman_to_int(s: str) -> int | None: - values = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100} + values = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100, "d": 500, "m": 1000} s = s.lower() if not s or any(c not in values for c in s): return None @@ -72,6 +66,49 @@ def _roman_to_int(s: str) -> int | None: return total or None +def _int_to_roman(n: int) -> str | None: + if not 1 <= n <= 39: + return None + out = [] + for val, sym in ((10, "X"), (9, "IX"), (5, "V"), (4, "IV"), (1, "I")): + while n >= val: + out.append(sym) + n -= val + return "".join(out) + + +# Spelled-out sequel indices, English + French, plus a few ordinals. +_NUMBER_WORDS: dict[str, int] = { + "one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "six": 6, "seven": 7, + "eight": 8, "nine": 9, "ten": 10, "eleven": 11, "twelve": 12, + "first": 1, "second": 2, "third": 3, + "un": 1, "deux": 2, "trois": 3, "quatre": 4, "cinq": 5, "sept": 7, + "huit": 8, "neuf": 9, "dix": 10, "onze": 11, "douze": 12, + "premier": 1, "première": 1, "deuxième": 2, "seconde": 2, "troisième": 3, +} +_YEAR_RE = re.compile(r"(?<!\d)(?:19|20)\d{2}(?!\d)") + + +def year_in(text: str) -> int | None: + """First 19xx/20xx in `text`, or None — used to lift a year off a show + folder name ("Some.Show.2022.S01") for the search fallback (§10.1/V8).""" + m = _YEAR_RE.search(text or "") + return int(m.group(0)) if m else None + + +def clean_query(s: str) -> str: + """ + Punctuation → spaces for a TMDB query, *without* the extension / + parenthesized-year stripping `naive_title` does. `naive_title` assumes + a real filename; a show's `display_title` is a folder basename + ("Some.Show.Name" — `rsplit('.', 1)` would eat ".Name"), so it needs a + gentler normaliser (§10.1/V8). + """ + s = re.sub(r"[._-]+", " ", s or "") + s = _strip_editions(s) + return re.sub(r"\s+", " ", s).strip() + + def _strip_editions(title: str) -> str: return re.sub(r"\s+", " ", _EDITION_RE.sub(" ", title)).strip() @@ -90,21 +127,50 @@ def naive_title(filename: str) -> str: return re.sub(r"\s+", " ", stem).strip() +_PART_KEYWORDS = r"part|chapter|volume|vol|partie|chapitre|volet|livre|book|episode" +_TRAILING_INDEX_RE = re.compile( + r"^(?P<base>.+?)(?:\s+(?:" + _PART_KEYWORDS + r"))?" + r"\s+(?P<num>\d{1,2}|[ivxlcdm]{1,6}|" + "|".join(_NUMBER_WORDS) + r")$", + re.IGNORECASE, +) + + +def _index_value(tok: str) -> int | None: + tok = tok.strip().lower() + if tok.isdigit(): + v = int(tok) + return v if 1 <= v <= 39 else None + if tok in _NUMBER_WORDS: + return _NUMBER_WORDS[tok] + return _roman_to_int(tok) + + def sequel_variants(title: str) -> list[str]: """ - A trailing sequel digit sometimes has no equivalent in the real TMDB - title, or the real title uses a Roman numeral instead (§3.3 row 4). - Returns extra candidates to try — empty if `title` has no trailing digit. + A trailing sequel index often has no exact match in the real TMDB + title: the file has a digit where TMDB uses a Roman numeral (or the + reverse), spells the number out, or wraps it as "Part N" / "Chapitre N" + (§3.3 row 4, §10.1/V10). Returns extra candidate titles to try — the + base alone, and the index re-rendered as digit and as Roman numeral. + Empty when `title` carries no recognisable trailing index. """ - m = re.match(r"^(.*\S)\s+([2-9])$", title) + m = _TRAILING_INDEX_RE.match(title.strip()) if not m: return [] - base, digit = m.group(1), int(m.group(2)) - variants = [base] - roman = _ROMAN_NUMERALS.get(digit) - if roman: - variants.append(f"{base} {roman}") - return variants + base = m.group("base").strip() + if len(base) < 2: + return [] + n = _index_value(m.group("num")) + if n is None: + return [] + tok = m.group("num").lower() + out = [base] + roman = _int_to_roman(n) + if roman and roman.lower() != tok: + out.append(f"{base} {roman}") + if str(n) != tok: + out.append(f"{base} {n}") + return [v for v in dict.fromkeys(out) if v != title] def season_from_folder_name(name: str) -> int | None: diff --git a/packages/meshbay-node/src/meshbay_node/media_cache.py b/packages/meshbay-node/src/meshbay_node/media_cache.py index 4600a09..8233270 100644 --- a/packages/meshbay-node/src/meshbay_node/media_cache.py +++ b/packages/meshbay-node/src/meshbay_node/media_cache.py @@ -175,6 +175,18 @@ class MediaCache: "(SELECT file_id FROM tmdb_override)", (file_id,)) await self._db.commit() + async def drop_tmdb_match(self, file_id: str) -> None: + """ + Full per-file reset: forget the match *and* any manual override + marker, so the next `media_meta_req` re-resolves from scratch with + the current matcher. This is the explicit operator "re-match this + one" action (§10.1/V13) — deliberately stronger than + `clear_file_tmdb`, which spares an override. + """ + await self._db.execute("DELETE FROM file_tmdb WHERE file_id = ?", (file_id,)) + await self._db.execute("DELETE FROM tmdb_override WHERE file_id = ?", (file_id,)) + await self._db.commit() + async def clear_tmdb_matches(self, file_ids: list[str]) -> int: """ Drop the auto-resolved file->tmdb mappings for these files so the diff --git a/packages/meshbay-node/src/meshbay_node/tmdb.py b/packages/meshbay-node/src/meshbay_node/tmdb.py index a530660..c9623df 100644 --- a/packages/meshbay-node/src/meshbay_node/tmdb.py +++ b/packages/meshbay-node/src/meshbay_node/tmdb.py @@ -51,24 +51,47 @@ def _normalize(s: str) -> str: return re.sub(r"\s+", " ", s).strip() -def _best_match(query_title: str, results: list[dict], keys: tuple[str, ...]) -> tuple[dict | None, float]: +def _release_year_of(item: dict) -> int | None: + d = str(item.get("release_date") or item.get("first_air_date") or "") + return int(d[:4]) if d[:4].isdigit() else None + + +def _ratio_against(qn: str, item: dict, keys: tuple[str, ...]) -> float: + best = 0.0 + for k in keys: + val = item.get(k) + if val: + best = max(best, difflib.SequenceMatcher(None, qn, _normalize(str(val))).ratio()) + return best + + +def _best_match( + query_title: str, results: list[dict], keys: tuple[str, ...], + year: int | None = None, +) -> tuple[dict | None, float]: """ Trusts TMDB's own ranking (§3.3's last row — a locally-recomputed - re-rank picked a coincidentally-closer-looking wrong show once): only - the top result is considered. The similarity ratio is returned purely - as a confidence signal for the caller's fallback decision, never used - to pick a different candidate. + re-rank picked a coincidentally-closer-looking wrong show once): the + top result is what's returned. The similarity ratio rides along purely + as a confidence signal for the caller's fallback decision. + + One narrow exception (§10.1/V9): when the top result is *not* a + confident textual hit (ratio < 0.6) and a `year` was requested, a + different result of that **exact** release year is preferred. TMDB + already year-filtered the search, so an entry landing on the requested + year is a hard corroboration, not the fuzzy string re-rank §3.3 warns + against — and this never overrides a confident top hit. """ if not results: return None, 0.0 - top = results[0] qn = _normalize(query_title) - best_ratio = 0.0 - for k in keys: - val = top.get(k) - if val: - best_ratio = max(best_ratio, difflib.SequenceMatcher(None, qn, _normalize(str(val))).ratio()) - return top, best_ratio + top = results[0] + top_ratio = _ratio_against(qn, top, keys) + if year is not None and top_ratio < 0.6: + for item in results: + if _release_year_of(item) == year: + return item, _ratio_against(qn, item, keys) + return top, top_ratio class TmdbClient: @@ -138,12 +161,15 @@ class TmdbClient: params["year"] = year data = await self._get("search/movie", params) results = (data or {}).get("results", []) - return _best_match(title, results, ("title", "original_title")) + return _best_match(title, results, ("title", "original_title"), year=year) - async def search_tv(self, title: str) -> tuple[dict | None, float]: - data = await self._get("search/tv", {"query": title}) + async def search_tv(self, title: str, year: int | None = None) -> tuple[dict | None, float]: + params = {"query": title} + if year: + params["first_air_date_year"] = year + data = await self._get("search/tv", params) results = (data or {}).get("results", []) - return _best_match(title, results, ("name", "original_name")) + return _best_match(title, results, ("name", "original_name"), year=year) async def search_movie_results(self, title: str) -> list[dict]: """ diff --git a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py index 788a8f1..af6bf08 100644 --- a/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py +++ b/packages/meshbay-node/src/meshbay_node/transport/webrtc_server.py @@ -70,6 +70,7 @@ from meshbay_common.adminop import ( OP_TMDB_ENABLED, OP_VIDEO_ROOT, OP_TMDB_OVERRIDE, + OP_TMDB_REMATCH, OP_MUSICBRAINZ_ENABLED, OP_AUDIO_ROOT, OP_PHOTO_ROOTS, @@ -475,6 +476,8 @@ class WebRTCPeerSession: self._spawn(self._do_tmdb_search_request(msg)) elif mtype == MNP.TMDB_OVERRIDE: self._do_tmdb_override(msg) + elif mtype == MNP.TMDB_REMATCH: + self._do_tmdb_rematch(msg) elif mtype == MNP.MUSICBRAINZ_ENABLED: self._do_musicbrainz_enabled(msg) elif mtype == MNP.MUSIC_META_REQ: @@ -3171,82 +3174,133 @@ class WebRTCPeerSession: except Exception: pass + def _do_tmdb_rematch(self, msg: dict) -> None: + """ + An operator dropping one file's cached TMDB match so it re-resolves + with the current matcher (§10.1/V13) — the one-click alternative to + the full search-and-pick "Fix match" flow, and reachable without + SSH (`meshbay-node video rematch` clears a whole group). Signed like + `tmdb_override`: `media_cache` is shared node-wide. + """ + file_id = msg.get("file_id") + if not isinstance(file_id, str) or not file_id: + self._send({"type": "error", "detail": "Missing file_id"}) + return + if not self._group_ctx()["index"].get_entry(file_id): + self._send({"type": "error", "detail": "File not found"}) + return + if not self._has_admin_authority(): + self._send({"type": "error", "detail": "No authorized key for this"}) + return + self._issue_admin_challenge(OP_TMDB_REMATCH, f"file_id={file_id}") + + async def _admin_exec_tmdb_rematch( + self, pending: dict, transcript: bytes, sig: bytes, + ) -> None: + subject = pending["subject"] + if not await self._verify_admin_sig(transcript, sig): + self._send({"type": "error", "detail": "Signature verification failed"}) + self._audit("admin_auth_failed", f"tmdb_rematch:{subject}") + return + file_id = dict(part.split("=", 1) for part in subject.split(","))["file_id"] + media_cache = self._ctx.get("media_cache") + if media_cache is None: + self._send({"type": "error", "detail": "Media cache not available"}) + return + await media_cache.drop_tmdb_match(file_id) + self._audit("tmdb_rematch", subject) + + notice = {"type": MNP.TMDB_REMATCH_ACK, "v": MNP_VERSION, "file_id": file_id} + for uid, session in list(self._peer_registry().items()): + try: + session._send(notice) + except Exception: + pass + async def _tmdb_search(self, tmdb_client, entry, is_show: bool): """ - §3.3's retry ladder. TMDB's own top result is still trusted per - query (§3.3's last row — no local re-ranking of *its* list); what - changed is that the ladder now *scores every candidate query* and - keeps the best, instead of returning the first that merely clears - 0.6. + §3.3's retry ladder — same shape for movies and shows (§10.1/V8). + TMDB's own top result is still trusted per query (§3.3's last row — + no local re-ranking of *its* list); what the ladder adds is that it + *scores every candidate query* and keeps the best, instead of + returning the first that merely clears 0.6. The bare parsed title is the weakest query: guessit drops a - "Volume 2", strips a real subtitle into `alternative_title`, and - renders a sequel number where TMDB uses a Roman numeral. A wrong - film that happened to score ~0.7 against that weak query — a - same-year making-of documentary, or a franchise entry whose - localized TMDB title *is* the franchise name — used to win outright - before `alternative_title` or the Roman-numeral variant was ever - tried. Found live (2026-08-29): a numbered sequel matched a - same-year documentary; a two-volume film's second part matched the - first; several franchise entries matched one early entry. + "Volume 2", strips a real subtitle into `alternative_title`, renders + a sequel index where TMDB spells it differently, and a show's folder + name can carry a year or release-group noise. A wrong entry that + scored ~0.7 against that weak query — a same-year making-of + documentary, a franchise entry whose localized TMDB title *is* the + franchise name, a season-specific promo entry standing in for a + whole show — used to win outright before a stronger candidate was + ever tried. Found live (2026-08-29). """ from meshbay_node.indexer import title_parse if is_show: title = entry.display_title or title_parse.naive_title(entry.name) - result, ratio = await tmdb_client.search_tv(title) - if result is None or ratio < 0.6: - naive = title_parse.naive_title(entry.name) - if naive != title: - result, ratio = await tmdb_client.search_tv(naive) - return result, ratio - - def _release_year(res: dict) -> int | None: - d = str(res.get("release_date") or res.get("first_air_date") or "") - return int(d[:4]) if d[:4].isdigit() else None + name_naive = title_parse.naive_title(entry.name) + year = title_parse.year_in(title) or title_parse.year_in(entry.name) + extra = [c for c in (name_naive, title_parse.clean_query(title)) + if c and c != title] + return await self._tmdb_ladder( + tmdb_client.search_tv, title, extra, year, strong_extra=False) parsed = title_parse.parse_movie_filename(entry.name) title = entry.display_title or parsed.display_title or parsed.naive_title + strong = [c for c in (parsed.alt_title, *title_parse.sequel_variants(title)) if c] + extra = [c for c in (*strong, parsed.naive_title) if c and c != title] + return await self._tmdb_ladder( + tmdb_client.search_movie, title, extra, parsed.year, + strong_extra=bool(strong)) - # Fast path, unchanged in effect: a strong direct hit still returns - # on the first call, so the common case costs exactly one request - # and the new ladder below only engages in the ambiguous 0<ratio<0.85 - # zone where every one of the live bugs lived. - result, ratio = await tmdb_client.search_movie(title, parsed.year) - if result is not None and ratio >= 0.85: - return result, ratio - - best_result, best_score = (result, ratio) if result is not None else (None, 0.0) + @staticmethod + async def _tmdb_ladder(search_fn, primary: str, extra: list[str], + year: int | None, strong_extra: bool): + """ + `search_fn(query, year) -> (result|None, ratio)`. Try `primary`, + return at once on a confident hit (ratio >= 0.85 — the common case, + one request). Otherwise score each `extra` candidate and keep the + best. `strong_extra` says whether `extra` contains anything more + specific than a punctuation-normalised restatement of `primary` + (an alternative_title, a sequel variant); when it does not and the + primary hit is already decent, the remaining calls are skipped + (§10.1/V11 — they almost never win and cost a round trip each). + """ + def _year_of(res: dict) -> int | None: + d = str(res.get("release_date") or res.get("first_air_date") or "") + return int(d[:4]) if d[:4].isdigit() else None - def _apply_year_rescue(res: dict, r: float) -> float: - # Only for a query whose own top hit is weak on its face - # (ratio < 0.6): TMDB already year-filtered the search, so its - # top result landing exactly on the filename's year is a hard - # corroborating signal that the low string ratio is a - # localized/rearranged title, not a wrong film. Never lets year - # equality outrank a genuinely strong textual match elsewhere. - if r < 0.6 and parsed.year and _release_year(res) == parsed.year: + def _rescue(res: dict, r: float) -> float: + # A sub-0.6 hit whose result lands on the exact requested year: + # TMDB already year-filtered the search, so this is a hard + # corroboration that the low ratio is a localised/rearranged + # title, not a wrong entry. Never overrides a confident hit. + if r < 0.6 and year and _year_of(res) == year: return max(r, 0.6) return r + result, ratio = await search_fn(primary, year) + if result is not None and ratio >= 0.85: + return result, ratio + best_result, best_score = (result, ratio) if result is not None else (None, 0.0) if best_result is not None: - best_score = _apply_year_rescue(best_result, best_score) + best_score = _rescue(best_result, best_score) + if best_score >= 0.6 and not strong_extra: + return best_result, best_score - for candidate in filter(None, [parsed.alt_title, - *title_parse.sequel_variants(title), - parsed.naive_title]): - if candidate == title: + for candidate in extra: + if candidate == primary: continue - r2, ratio2 = await tmdb_client.search_movie(candidate, parsed.year) - if r2 is None and parsed.year: - # A year-filtered search that finds nothing: the filename's - # year tag may be an edition/regional year TMDB doesn't - # carry. Retry the same candidate unconstrained before - # dropping it. - r2, ratio2 = await tmdb_client.search_movie(candidate) + r2, ratio2 = await search_fn(candidate, year) + if r2 is None and year: + # A year-filtered search that finds nothing: the year tag + # may be an edition/regional year TMDB doesn't carry. Retry + # the candidate unconstrained before dropping it. + r2, ratio2 = await search_fn(candidate, None) if r2 is None: continue - score2 = _apply_year_rescue(r2, ratio2) + score2 = _rescue(r2, ratio2) if score2 > best_score: best_result, best_score = r2, score2 if best_score >= 0.85: @@ -3866,6 +3920,9 @@ class WebRTCPeerSession: elif pending["op"] == OP_TMDB_OVERRIDE: self._spawn( self._admin_exec_tmdb_override(pending, transcript, sig_bytes)) + elif pending["op"] == OP_TMDB_REMATCH: + self._spawn( + self._admin_exec_tmdb_rematch(pending, transcript, sig_bytes)) elif pending["op"] == OP_MUSICBRAINZ_ENABLED: self._spawn( self._admin_exec_musicbrainz_enabled(pending, transcript, sig_bytes)) diff --git a/packages/meshbay-node/tests/test_media_cache.py b/packages/meshbay-node/tests/test_media_cache.py index 4e65b24..f12a366 100644 --- a/packages/meshbay-node/tests/test_media_cache.py +++ b/packages/meshbay-node/tests/test_media_cache.py @@ -107,6 +107,19 @@ async def test_clear_file_tmdb_drops_one_auto_match_but_keeps_an_override(cache) @pytest.mark.asyncio +async def test_drop_tmdb_match_forgets_both_the_match_and_the_override(cache): + await cache.set_file_tmdb("f", "10", "movie") + await cache.mark_tmdb_override("f") + + await cache.drop_tmdb_match("f") + + assert await cache.get_file_tmdb("f") is None + # override marker is gone: a fresh match is now treated as ordinary + await cache.set_file_tmdb("f", "20", "movie") + assert await cache.clear_tmdb_matches(["f"]) == 1 + + +@pytest.mark.asyncio async def test_prune_file_also_clears_the_override_marker(cache): await cache.set_file_tmdb("gone", "10", "movie") await cache.mark_tmdb_override("gone") diff --git a/packages/meshbay-node/tests/test_title_parse.py b/packages/meshbay-node/tests/test_title_parse.py index 03fb588..ac4d434 100644 --- a/packages/meshbay-node/tests/test_title_parse.py +++ b/packages/meshbay-node/tests/test_title_parse.py @@ -6,12 +6,14 @@ is a manual acceptance step (§11), not something this repo's corpus holds. from meshbay_node.indexer.title_parse import ( ParsedName, + clean_query, leading_episode_number, naive_title, parse_episode_filename, parse_movie_filename, season_from_folder_name, sequel_variants, + year_in, ) @@ -69,6 +71,48 @@ def test_sequel_variants_empty_when_no_trailing_digit(): assert sequel_variants("Some Movie") == [] +# ── §10.1/V10: wider sequel-index handling ────────────────────────────────── + +def test_sequel_variants_roman_numeral_offers_the_digit_form(): + v = sequel_variants("Old Frontier III") + assert "Old Frontier" in v + assert "Old Frontier 3" in v + + +def test_sequel_variants_strips_a_part_keyword_wrapper(): + v = sequel_variants("Some Saga Part 2") + assert "Some Saga" in v + assert "Some Saga II" in v + + +def test_sequel_variants_reads_a_spelled_out_index(): + v = sequel_variants("Story Chapter Three") + assert "Story" in v + assert "Story 3" in v and "Story III" in v + + +def test_sequel_variants_ignores_a_trailing_word_that_is_not_an_index(): + assert sequel_variants("The Dark Knight") == [] + assert sequel_variants("In Bruges") == [] + + +def test_sequel_variants_ignores_a_four_digit_year_suffix(): + assert sequel_variants("Blade Runner 2049") == [] + + +def test_year_in_lifts_a_year_from_a_show_folder_name(): + assert year_in("Some.Show.2022.S01") == 2022 + assert year_in("Some Show") is None + assert year_in("Episode 100 of 2010") == 2010 + + +def test_clean_query_despaces_a_folder_name_without_eating_the_last_word(): + # naive_title would rsplit on the last dot and drop ".Name" + assert clean_query("Some.Show.Name") == "Some Show Name" + assert clean_query("Some.Show.Name.S01") == "Some Show Name S01" + assert naive_title("Some.Show.Name") != "Some Show Name" # the trap it avoids + + # ── bug 2026-08-29: guessit peels "Volume N" off the title ───────────────── def test_movie_volume_number_is_folded_back_into_the_title(): diff --git a/packages/meshbay-node/tests/test_tmdb.py b/packages/meshbay-node/tests/test_tmdb.py index fcbc2a2..ab452ce 100644 --- a/packages/meshbay-node/tests/test_tmdb.py +++ b/packages/meshbay-node/tests/test_tmdb.py @@ -40,6 +40,59 @@ async def test_search_movie_returns_top_result_and_confidence(): await client.close() +# ── §10.1/V9: year-exact preference, only when the top hit is weak ────────── + +@pytest.mark.asyncio +async def test_year_exact_result_wins_when_the_top_hit_is_low_confidence(): + # results[0] is TMDB's popularity #1 but a poor textual match for the + # query; results[2] is the exact requested year. + body = {"results": [ + {"id": 1, "title": "Franchise vs. The Doctor", "release_date": "1962-10-05"}, + {"id": 2, "title": "Franchise: Goldfinger", "release_date": "1964-09-17"}, + {"id": 3, "title": "Second Errand", "release_date": "2002-11-20"}, + ]} + client = TmdbClient( + roster=FakeRoster(), + transport=httpx.MockTransport(_handler({"search/movie": body})), + ) + result, _ = await client.search_movie("The Franchise", 2002) + + assert result["id"] == 3 + await client.close() + + +@pytest.mark.asyncio +async def test_year_is_ignored_when_the_top_hit_is_already_confident(): + body = {"results": [ + {"id": 1, "title": "The Franchise", "release_date": "1999-01-01"}, + {"id": 2, "title": "Unrelated", "release_date": "2002-01-01"}, + ]} + client = TmdbClient( + roster=FakeRoster(), + transport=httpx.MockTransport(_handler({"search/movie": body})), + ) + result, ratio = await client.search_movie("The Franchise", 2002) + + assert result["id"] == 1 and ratio > 0.9 + await client.close() + + +@pytest.mark.asyncio +async def test_no_year_match_leaves_the_top_result_in_place(): + body = {"results": [ + {"id": 1, "title": "Something Else Entirely", "release_date": "1990-01-01"}, + {"id": 2, "title": "Also Not It", "release_date": "1991-01-01"}, + ]} + client = TmdbClient( + roster=FakeRoster(), + transport=httpx.MockTransport(_handler({"search/movie": body})), + ) + result, _ = await client.search_movie("The Franchise", 2002) + + assert result["id"] == 1 + await client.close() + + @pytest.mark.asyncio async def test_search_tv_returns_top_result(): body = {"results": [{"id": 7, "name": "Some Show"}]} diff --git a/packages/meshbay-node/tests/test_tmdb_rematch_policy.py b/packages/meshbay-node/tests/test_tmdb_rematch_policy.py new file mode 100644 index 0000000..ff259c1 --- /dev/null +++ b/packages/meshbay-node/tests/test_tmdb_rematch_policy.py @@ -0,0 +1,147 @@ +""" +`tmdb_rematch` (§10.1/V13) — an operator dropping one file's cached TMDB +match so it re-resolves with the current matcher. Signed like +`tmdb_override` (media_cache is shared node-wide); unlike `clear_file_tmdb` +it forgets a manual override marker too, since the operator is explicitly +asking for a fresh resolution. +""" + +import hashlib + +import pytest +from conftest import one_root +from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey +from meshbay_common.adminop import OP_TMDB_REMATCH +from meshbay_common.protocol import MNP, IndexEntry +from meshbay_node.indexer.group_index import GroupIndex +from meshbay_node.media_cache import MediaCache +from meshbay_node.transport.webrtc_server import WebRTCPeerSession + +pytestmark = pytest.mark.asyncio + + +def _session(tmp_path, user_id, *, operator=None): + shared = tmp_path / "shared" + shared.mkdir(exist_ok=True) + index = GroupIndex(group_id="g" * 32, sk_node=Ed25519PrivateKey.generate()) + s = WebRTCPeerSession.__new__(WebRTCPeerSession) + s._ctx = {"roots": one_root(shared), "index": index, "sk_node": index.sk_node, + "node_user_id": operator} + s._group_id = None + s._user_id = user_id + s._pk_user = "" + s.sent = [] + s._send = s.sent.append + s._audit = lambda *a, **k: None + return s + + +def _entry(name): + digest = hashlib.sha256(name.encode()).hexdigest() + return IndexEntry(id=digest, name=name, path="movies", size=1, type="video", + added_at=0, display_title="Some Film") + + +async def _true(): + return True + + +async def test_missing_file_id_is_refused(tmp_path): + s = _session(tmp_path, "op", operator="op") + s._has_admin_authority = lambda: True + issued = [] + s._issue_admin_challenge = lambda op, subject: issued.append((op, subject)) + + s._do_tmdb_rematch({}) + + assert not issued + assert [m for m in s.sent if m.get("type") == "error"] + + +async def test_unknown_file_id_is_refused(tmp_path): + s = _session(tmp_path, "op", operator="op") + s._has_admin_authority = lambda: True + issued = [] + s._issue_admin_challenge = lambda op, subject: issued.append((op, subject)) + + s._do_tmdb_rematch({"file_id": "nope"}) + + assert not issued + assert [m for m in s.sent if m.get("type") == "error"] + + +async def test_no_authorized_key_is_refused(tmp_path): + s = _session(tmp_path, "member", operator="the-operator") + e = _entry("some.film.2001.mkv") + s._ctx["index"].add_entry(e) + s._has_admin_authority = lambda: False + + s._do_tmdb_rematch({"file_id": e.id}) + + assert [m for m in s.sent if m.get("type") == "error"] + + +async def test_a_valid_request_is_signed(tmp_path): + s = _session(tmp_path, "op", operator="op") + e = _entry("some.film.2001.mkv") + s._ctx["index"].add_entry(e) + s._has_admin_authority = lambda: True + issued = [] + s._issue_admin_challenge = lambda op, subject: issued.append((op, subject)) + + s._do_tmdb_rematch({"file_id": e.id}) + + assert issued == [(OP_TMDB_REMATCH, f"file_id={e.id}")] + + +async def test_exec_drops_the_match_and_the_override_marker(tmp_path): + s = _session(tmp_path, "op", operator="op") + e = _entry("some.film.2001.mkv") + s._ctx["index"].add_entry(e) + + media_cache = MediaCache(db_path=tmp_path / "media_cache.db") + await media_cache.open() + try: + s._ctx["media_cache"] = media_cache + await media_cache.set_file_tmdb(e.id, "wrong-id", "movie") + await media_cache.mark_tmdb_override(e.id) + s._verify_admin_sig = lambda transcript, sig: _true() + peer = type("Peer", (), {"sent": []})() + peer._send = peer.sent.append + s._peer_registry = lambda: {"peer-1": peer} + + await s._admin_exec_tmdb_rematch( + {"subject": f"file_id={e.id}"}, b"transcript", b"sig") + + assert await media_cache.get_file_tmdb(e.id) is None + # the override marker is gone too, so a later group-wide rematch + # would treat a fresh match as ordinary + await media_cache.set_file_tmdb(e.id, "fresh", "movie") + assert await media_cache.clear_tmdb_matches([e.id]) == 1 + assert [m for m in peer.sent if m.get("type") == MNP.TMDB_REMATCH_ACK] + finally: + await media_cache.close() + + +async def test_exec_refuses_a_bad_signature(tmp_path): + s = _session(tmp_path, "op", operator="op") + e = _entry("some.film.2001.mkv") + s._ctx["index"].add_entry(e) + + media_cache = MediaCache(db_path=tmp_path / "media_cache.db") + await media_cache.open() + try: + s._ctx["media_cache"] = media_cache + await media_cache.set_file_tmdb(e.id, "keep-me", "movie") + + async def _false(): + return False + s._verify_admin_sig = lambda transcript, sig: _false() + + await s._admin_exec_tmdb_rematch( + {"subject": f"file_id={e.id}"}, b"transcript", b"badsig") + + assert await media_cache.get_file_tmdb(e.id) == ("keep-me", "movie") + assert [m for m in s.sent if m.get("type") == "error"] + finally: + await media_cache.close() diff --git a/packages/meshbay-node/tests/test_tmdb_search_ladder.py b/packages/meshbay-node/tests/test_tmdb_search_ladder.py index 49b71bd..9865c77 100644 --- a/packages/meshbay-node/tests/test_tmdb_search_ladder.py +++ b/packages/meshbay-node/tests/test_tmdb_search_ladder.py @@ -61,8 +61,12 @@ class LadderTmdb: return self._table[(title, None)] return None, 0.0 - async def search_tv(self, title): - self.calls.append(("tv", title)) + async def search_tv(self, title, year=None): + self.calls.append((title, year)) + if (title, year) in self._table: + return self._table[(title, year)] + if (title, None) in self._table: + return self._table[(title, None)] return None, 0.0 @@ -80,6 +84,17 @@ async def _run(name: str, table: dict): return result, ratio, client +async def _run_show(display_title: str, name: str, table: dict): + entry = IndexEntry( + id="s", name=name, path="Show/S1", size=1, type="video", added_at=0, + display_title=display_title, season=1, episode=1, + ) + client = LadderTmdb(table) + session = WebRTCPeerSession.__new__(WebRTCPeerSession) + result, ratio = await session._tmdb_search(client, entry, is_show=True) + return result, ratio, client + + # ── a two-part film: the parts must resolve to different entries ──────────── async def test_second_volume_resolves_to_the_second_volume(): @@ -191,3 +206,42 @@ async def test_strong_direct_match_costs_a_single_request(): _, _, client = await _run("A.Quiet.Film.2010.mkv", {("A Quiet Film", 2010): (STANDALONE, 1.0)}) assert client.calls == [("A Quiet Film", 2010)] + + +# ── §10.1/V11: a decent primary hit with nothing more specific to try ────── + +async def test_decent_primary_with_no_stronger_candidate_costs_one_call(): + result, _, client = await _run( + "A.Quiet.Film.2010.mkv", + {("A Quiet Film", 2010): ({"id": "77", "title": "A Quiet Movie", + "release_date": "2010-01-01"}, 0.70)}, + ) + assert result["id"] == "77" + assert client.calls == [("A Quiet Film", 2010)], ( + "no alt title, no sequel index → the punctuation-restatement fallback " + "is not worth a second request once the primary hit is decent") + + +# ── §10.1/V8: the show branch uses the same scored ladder ───────────────── + +async def test_show_scored_ladder_beats_a_weak_primary_hit(): + result, _, _ = await _run_show( + "Some.Show.Name.S01", "s01e01.mkv", + { + ("Some.Show.Name.S01", None): ({"id": "10", "name": "Promo Special", + "first_air_date": "2019-01-01"}, 0.55), + ("Some Show Name S01", None): ({"id": "20", "name": "Some Show Name", + "first_air_date": "2017-01-01"}, 0.95), + }, + ) + assert result["id"] == "20" + + +async def test_show_year_in_folder_name_rescues_a_weak_hit(): + result, ratio, _ = await _run_show( + "Some.Show.2019.S01", "s01e01.mkv", + {("Some.Show.2019.S01", 2019): ({"id": "30", "name": "Some Show", + "first_air_date": "2019-05-01"}, 0.30)}, + ) + assert result["id"] == "30" + assert ratio >= 0.6 |