From 5d28d0c96cc8489645178b483835489779cb3887 Mon Sep 17 00:00:00 2001 From: Christophe Besson Date: Sat, 29 Aug 2026 15:43:38 +0200 Subject: feat(node): V8–V11 — show-branch ladder, year-aware _best_match, wider sequel_variants MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit V8: the TV/show branch of _tmdb_search used the old "first candidate over 0.6 wins" shape. It now shares one _tmdb_ladder helper with the movie branch — score every candidate query, keep the best, fast-path a confident primary hit. A year lifted off the show's folder name (title_parse.year_in, e.g. "Some.Show.2022.S01") rescues a sub-0.6 hit that lands on the exact year. title_parse.clean_query de-dots a folder-derived title without naive_title's extension-stripping trap. V9: _best_match gains an optional `year`. When the top result is not a confident textual hit (ratio < 0.6) and a year was requested, a different result of that exact release year is preferred — TMDB already year-filtered the search, so this is a hard corroboration, not the fuzzy re-rank §3.3 warns against. A confident top hit is never overridden. search_movie/search_tv forward the year. V10: sequel_variants widened — trailing Roman→digit as well as digit→Roman, spelled-out indices (one..twelve / un..douze / ordinals), and a "Part N" / "Chapitre N" wrapper. Still empty for a trailing word that is not an index or a 4-digit year. V11: when the primary hit is already decent (>= 0.6) and there is nothing more specific to try (no alternative_title, no sequel variant — only a punctuation restatement left), the ladder returns without the extra requests. The clean-title common case is back to one call. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_018BMLQjqFGCize2KtNBT79v --- .../src/meshbay_node/indexer/title_parse.py | 100 +++++++++++++++++---- 1 file changed, 83 insertions(+), 17 deletions(-) (limited to 'packages/meshbay-node/src/meshbay_node/indexer') diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 24c23bc..018746b 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -52,14 +52,8 @@ _SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE) # some other word that merely starts with "s" followed by digits. _SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE) -_ROMAN_NUMERALS = { - 2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI", - 7: "VII", 8: "VIII", 9: "IX", 10: "X", -} - - def _roman_to_int(s: str) -> int | None: - values = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100} + values = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100, "d": 500, "m": 1000} s = s.lower() if not s or any(c not in values for c in s): return None @@ -72,6 +66,49 @@ def _roman_to_int(s: str) -> int | None: return total or None +def _int_to_roman(n: int) -> str | None: + if not 1 <= n <= 39: + return None + out = [] + for val, sym in ((10, "X"), (9, "IX"), (5, "V"), (4, "IV"), (1, "I")): + while n >= val: + out.append(sym) + n -= val + return "".join(out) + + +# Spelled-out sequel indices, English + French, plus a few ordinals. +_NUMBER_WORDS: dict[str, int] = { + "one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "six": 6, "seven": 7, + "eight": 8, "nine": 9, "ten": 10, "eleven": 11, "twelve": 12, + "first": 1, "second": 2, "third": 3, + "un": 1, "deux": 2, "trois": 3, "quatre": 4, "cinq": 5, "sept": 7, + "huit": 8, "neuf": 9, "dix": 10, "onze": 11, "douze": 12, + "premier": 1, "première": 1, "deuxième": 2, "seconde": 2, "troisième": 3, +} +_YEAR_RE = re.compile(r"(? int | None: + """First 19xx/20xx in `text`, or None — used to lift a year off a show + folder name ("Some.Show.2022.S01") for the search fallback (§10.1/V8).""" + m = _YEAR_RE.search(text or "") + return int(m.group(0)) if m else None + + +def clean_query(s: str) -> str: + """ + Punctuation → spaces for a TMDB query, *without* the extension / + parenthesized-year stripping `naive_title` does. `naive_title` assumes + a real filename; a show's `display_title` is a folder basename + ("Some.Show.Name" — `rsplit('.', 1)` would eat ".Name"), so it needs a + gentler normaliser (§10.1/V8). + """ + s = re.sub(r"[._-]+", " ", s or "") + s = _strip_editions(s) + return re.sub(r"\s+", " ", s).strip() + + def _strip_editions(title: str) -> str: return re.sub(r"\s+", " ", _EDITION_RE.sub(" ", title)).strip() @@ -90,21 +127,50 @@ def naive_title(filename: str) -> str: return re.sub(r"\s+", " ", stem).strip() +_PART_KEYWORDS = r"part|chapter|volume|vol|partie|chapitre|volet|livre|book|episode" +_TRAILING_INDEX_RE = re.compile( + r"^(?P.+?)(?:\s+(?:" + _PART_KEYWORDS + r"))?" + r"\s+(?P\d{1,2}|[ivxlcdm]{1,6}|" + "|".join(_NUMBER_WORDS) + r")$", + re.IGNORECASE, +) + + +def _index_value(tok: str) -> int | None: + tok = tok.strip().lower() + if tok.isdigit(): + v = int(tok) + return v if 1 <= v <= 39 else None + if tok in _NUMBER_WORDS: + return _NUMBER_WORDS[tok] + return _roman_to_int(tok) + + def sequel_variants(title: str) -> list[str]: """ - A trailing sequel digit sometimes has no equivalent in the real TMDB - title, or the real title uses a Roman numeral instead (§3.3 row 4). - Returns extra candidates to try — empty if `title` has no trailing digit. + A trailing sequel index often has no exact match in the real TMDB + title: the file has a digit where TMDB uses a Roman numeral (or the + reverse), spells the number out, or wraps it as "Part N" / "Chapitre N" + (§3.3 row 4, §10.1/V10). Returns extra candidate titles to try — the + base alone, and the index re-rendered as digit and as Roman numeral. + Empty when `title` carries no recognisable trailing index. """ - m = re.match(r"^(.*\S)\s+([2-9])$", title) + m = _TRAILING_INDEX_RE.match(title.strip()) if not m: return [] - base, digit = m.group(1), int(m.group(2)) - variants = [base] - roman = _ROMAN_NUMERALS.get(digit) - if roman: - variants.append(f"{base} {roman}") - return variants + base = m.group("base").strip() + if len(base) < 2: + return [] + n = _index_value(m.group("num")) + if n is None: + return [] + tok = m.group("num").lower() + out = [base] + roman = _int_to_roman(n) + if roman and roman.lower() != tok: + out.append(f"{base} {roman}") + if str(n) != tok: + out.append(f"{base} {n}") + return [v for v in dict.fromkeys(out) if v != title] def season_from_folder_name(name: str) -> int | None: -- cgit v1.2.3