diff options
| author | Christophe Besson <cbesson@gmail.com> | 2026-08-29 15:43:38 +0200 |
|---|---|---|
| committer | Christophe Besson <cbesson@gmail.com> | 2026-08-29 15:43:38 +0200 |
| commit | 5d28d0c96cc8489645178b483835489779cb3887 (patch) | |
| tree | 90248f1c3211ca37495a774d0bc140109c299abc /packages/meshbay-node/src/meshbay_node/indexer | |
| parent | 236e5e811355212945b89c4f7a99df5c837e97c7 (diff) | |
| download | meshbay-5d28d0c96cc8489645178b483835489779cb3887.tar.gz | |
feat(node): V8–V11 — show-branch ladder, year-aware _best_match, wider sequel_variants
V8: the TV/show branch of _tmdb_search used the old "first candidate over
0.6 wins" shape. It now shares one _tmdb_ladder helper with the movie
branch — score every candidate query, keep the best, fast-path a
confident primary hit. A year lifted off the show's folder name
(title_parse.year_in, e.g. "Some.Show.2022.S01") rescues a sub-0.6 hit
that lands on the exact year. title_parse.clean_query de-dots a
folder-derived title without naive_title's extension-stripping trap.
V9: _best_match gains an optional `year`. When the top result is not a
confident textual hit (ratio < 0.6) and a year was requested, a
different result of that exact release year is preferred — TMDB already
year-filtered the search, so this is a hard corroboration, not the fuzzy
re-rank §3.3 warns against. A confident top hit is never overridden.
search_movie/search_tv forward the year.
V10: sequel_variants widened — trailing Roman→digit as well as
digit→Roman, spelled-out indices (one..twelve / un..douze / ordinals),
and a "Part N" / "Chapitre N" wrapper. Still empty for a trailing word
that is not an index or a 4-digit year.
V11: when the primary hit is already decent (>= 0.6) and there is nothing
more specific to try (no alternative_title, no sequel variant — only a
punctuation restatement left), the ladder returns without the extra
requests. The clean-title common case is back to one call.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018BMLQjqFGCize2KtNBT79v
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node/indexer')
| -rw-r--r-- | packages/meshbay-node/src/meshbay_node/indexer/title_parse.py | 100 |
1 files changed, 83 insertions, 17 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py index 24c23bc..018746b 100644 --- a/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py +++ b/packages/meshbay-node/src/meshbay_node/indexer/title_parse.py @@ -52,14 +52,8 @@ _SPECIALS_RE = re.compile(r"\b(?:bonus|extras?|specials?)\b", re.IGNORECASE) # some other word that merely starts with "s" followed by digits. _SEASON_ABBREV_RE = re.compile(r"^s(\d{1,2})$", re.IGNORECASE) -_ROMAN_NUMERALS = { - 2: "II", 3: "III", 4: "IV", 5: "V", 6: "VI", - 7: "VII", 8: "VIII", 9: "IX", 10: "X", -} - - def _roman_to_int(s: str) -> int | None: - values = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100} + values = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100, "d": 500, "m": 1000} s = s.lower() if not s or any(c not in values for c in s): return None @@ -72,6 +66,49 @@ def _roman_to_int(s: str) -> int | None: return total or None +def _int_to_roman(n: int) -> str | None: + if not 1 <= n <= 39: + return None + out = [] + for val, sym in ((10, "X"), (9, "IX"), (5, "V"), (4, "IV"), (1, "I")): + while n >= val: + out.append(sym) + n -= val + return "".join(out) + + +# Spelled-out sequel indices, English + French, plus a few ordinals. +_NUMBER_WORDS: dict[str, int] = { + "one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "six": 6, "seven": 7, + "eight": 8, "nine": 9, "ten": 10, "eleven": 11, "twelve": 12, + "first": 1, "second": 2, "third": 3, + "un": 1, "deux": 2, "trois": 3, "quatre": 4, "cinq": 5, "sept": 7, + "huit": 8, "neuf": 9, "dix": 10, "onze": 11, "douze": 12, + "premier": 1, "première": 1, "deuxième": 2, "seconde": 2, "troisième": 3, +} +_YEAR_RE = re.compile(r"(?<!\d)(?:19|20)\d{2}(?!\d)") + + +def year_in(text: str) -> int | None: + """First 19xx/20xx in `text`, or None — used to lift a year off a show + folder name ("Some.Show.2022.S01") for the search fallback (§10.1/V8).""" + m = _YEAR_RE.search(text or "") + return int(m.group(0)) if m else None + + +def clean_query(s: str) -> str: + """ + Punctuation → spaces for a TMDB query, *without* the extension / + parenthesized-year stripping `naive_title` does. `naive_title` assumes + a real filename; a show's `display_title` is a folder basename + ("Some.Show.Name" — `rsplit('.', 1)` would eat ".Name"), so it needs a + gentler normaliser (§10.1/V8). + """ + s = re.sub(r"[._-]+", " ", s or "") + s = _strip_editions(s) + return re.sub(r"\s+", " ", s).strip() + + def _strip_editions(title: str) -> str: return re.sub(r"\s+", " ", _EDITION_RE.sub(" ", title)).strip() @@ -90,21 +127,50 @@ def naive_title(filename: str) -> str: return re.sub(r"\s+", " ", stem).strip() +_PART_KEYWORDS = r"part|chapter|volume|vol|partie|chapitre|volet|livre|book|episode" +_TRAILING_INDEX_RE = re.compile( + r"^(?P<base>.+?)(?:\s+(?:" + _PART_KEYWORDS + r"))?" + r"\s+(?P<num>\d{1,2}|[ivxlcdm]{1,6}|" + "|".join(_NUMBER_WORDS) + r")$", + re.IGNORECASE, +) + + +def _index_value(tok: str) -> int | None: + tok = tok.strip().lower() + if tok.isdigit(): + v = int(tok) + return v if 1 <= v <= 39 else None + if tok in _NUMBER_WORDS: + return _NUMBER_WORDS[tok] + return _roman_to_int(tok) + + def sequel_variants(title: str) -> list[str]: """ - A trailing sequel digit sometimes has no equivalent in the real TMDB - title, or the real title uses a Roman numeral instead (§3.3 row 4). - Returns extra candidates to try — empty if `title` has no trailing digit. + A trailing sequel index often has no exact match in the real TMDB + title: the file has a digit where TMDB uses a Roman numeral (or the + reverse), spells the number out, or wraps it as "Part N" / "Chapitre N" + (§3.3 row 4, §10.1/V10). Returns extra candidate titles to try — the + base alone, and the index re-rendered as digit and as Roman numeral. + Empty when `title` carries no recognisable trailing index. """ - m = re.match(r"^(.*\S)\s+([2-9])$", title) + m = _TRAILING_INDEX_RE.match(title.strip()) if not m: return [] - base, digit = m.group(1), int(m.group(2)) - variants = [base] - roman = _ROMAN_NUMERALS.get(digit) - if roman: - variants.append(f"{base} {roman}") - return variants + base = m.group("base").strip() + if len(base) < 2: + return [] + n = _index_value(m.group("num")) + if n is None: + return [] + tok = m.group("num").lower() + out = [base] + roman = _int_to_roman(n) + if roman and roman.lower() != tok: + out.append(f"{base} {roman}") + if str(n) != tok: + out.append(f"{base} {n}") + return [v for v in dict.fromkeys(out) if v != title] def season_from_folder_name(name: str) -> int | None: |