summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-node/src/meshbay_node
diff options
context:
space:
mode:
authorChristophe Besson <cbesson@gmail.com>2026-08-25 14:50:36 +0200
committerChristophe Besson <cbesson@gmail.com>2026-08-25 14:50:36 +0200
commitbbc76ad4da3ca61f820b9605ec9228c7a9357352 (patch)
treee1295931bfdfcaa28174c247ff9dd37627022f9d /packages/meshbay-node/src/meshbay_node
parentc52e85b984bffc78125aff744513bc5d33ad72b3 (diff)
downloadmeshbay-bbc76ad4da3ca61f820b9605ec9228c7a9357352.tar.gz
fix(node): fall back to a loose MusicBrainz query when the strict one finds nothing
search_release() only ever tried a field-scoped exact-phrase Lucene query (artist:"..." AND release:"..."). Verified live against musicbrainz.org: any deviation from MusicBrainz's own spelling (a year suffix, an edition tag, an artist credited under an older/aliased name) drops it to zero results outright rather than a low-scored one, so the confidence check never even ran — this is why well-recognized artists were still getting almost no cover art. Add an unscoped loose-query fallback, escape Lucene special characters in the interpolated tag text, and make the confidence score consider the artist match too (not just the album title) now that the fallback has no field scoping to rely on. Also document MESHBAY_MUSICBRAINZ_CONTACT_DEFAULT in the systemd units and man page, mirroring MESHBAY_TMDB_DEFAULT_TOKEN's precedent — never a literal value in source, configured via node.env. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013XSohfUQQiaE77qyFLgSv3
Diffstat (limited to 'packages/meshbay-node/src/meshbay_node')
-rw-r--r--packages/meshbay-node/src/meshbay_node/musicbrainz.py74
1 files changed, 63 insertions, 11 deletions
diff --git a/packages/meshbay-node/src/meshbay_node/musicbrainz.py b/packages/meshbay-node/src/meshbay_node/musicbrainz.py
index f3d9b4f..6ca1cf4 100644
--- a/packages/meshbay-node/src/meshbay_node/musicbrainz.py
+++ b/packages/meshbay-node/src/meshbay_node/musicbrainz.py
@@ -64,23 +64,49 @@ def _normalize(s: str) -> str:
return re.sub(r"\s+", " ", s).strip()
-def _best_match(query: str, results: list[dict], key: str) -> tuple[dict | None, float]:
+# Lucene/Solr query-syntax characters (the parser MusicBrainz's `ws/2` search
+# runs on). A tag or folder-derived artist/album string is untrusted free
+# text as far as this parser is concerned — a stray "(", ":" or bare '"'
+# either breaks the surrounding quoted phrase or gets read as field/grouping
+# syntax rather than a literal character. Backslash-escaping each one keeps
+# it literal without changing what the analyzer tokenizes.
+_LUCENE_SPECIAL_RE = re.compile(r'([+\-&|!(){}\[\]^"~*?:\\/])')
+
+
+def _escape_lucene(s: str) -> str:
+ return _LUCENE_SPECIAL_RE.sub(r"\\\1", s)
+
+
+def _similarity(query: str, val: str | None) -> float:
+ return (difflib.SequenceMatcher(None, _normalize(query), _normalize(str(val))).ratio()
+ if val else 0.0)
+
+
+def _best_match_release(artist: str, album: str, results: list[dict]) -> tuple[dict | None, float]:
"""
Same "trust the search's own ranking" shape as tmdb.py's `_best_match`
(§3.3 of mediacenter.md found a locally-recomputed re-rank pick a
coincidentally closer-looking wrong result once — no reason to expect
MusicBrainz's own scored search to fare differently under the same
treatment). MusicBrainz already returns results ordered by its own
- `score`; only the top one is considered, and the similarity ratio is a
- confidence signal for the caller's fallback decision, not a re-rank.
+ `score`; only the top one is considered.
+
+ Confidence is the average of the album/title and artist similarity,
+ not the title alone: `search_release`'s loose fallback query has no
+ field scoping at all, so a title-only ratio would happily call a
+ same-titled album by an unrelated artist a good match. Averaging both
+ still lets a strong single-field match (e.g. the artist matches
+ exactly but the local album string carries an edition suffix) clear
+ the caller's threshold, while an unrelated same-name result does not.
"""
if not results:
return None, 0.0
top = results[0]
- val = top.get(key)
- ratio = (difflib.SequenceMatcher(None, _normalize(query), _normalize(str(val))).ratio()
- if val else 0.0)
- return top, ratio
+ title_ratio = _similarity(album, top.get("title"))
+ credit = top.get("artist-credit") or []
+ artist_name = credit[0].get("name") if credit else None
+ artist_ratio = _similarity(artist, artist_name)
+ return top, (title_ratio + artist_ratio) / 2
class MusicBrainzClient:
@@ -132,11 +158,37 @@ class MusicBrainzClient:
return None
async def search_release(self, artist: str, album: str) -> tuple[dict | None, float]:
- """Top MusicBrainz release match for an (artist, album) pair."""
- query = f'artist:"{artist}" AND release:"{album}"'
- data = await self._get(_BASE_URL + "release", {"query": query})
+ """
+ Top MusicBrainz release match for an (artist, album) pair.
+
+ Tries a field-scoped exact-phrase query first — cheap, and the
+ common case since tag data usually already matches MusicBrainz's
+ own spelling. Verified live against the real service: real tag/
+ folder text routinely carries something an exact phrase does not
+ tolerate at all — a year suffix, an edition tag, a stray
+ punctuation mark, or an artist credited under an older/aliased
+ name on that specific release — and MusicBrainz's phrase parser
+ then returns **zero** hits, not a low-scored one, so the caller's
+ confidence check never gets a chance to run at all.
+ `artist:"Groundation" AND release:"Hebron Gate"` finds it, score
+ 100; appending "(2003)" — exactly how many rip folders name a
+ release — drops it to zero results outright. A second, unscoped,
+ unquoted query lets MusicBrainz's own relevance ranking find the
+ same release regardless of the extra text — confirmed with the
+ same pair, buried among 6000+ candidates by the plain query, still
+ ranked first. Only tried when the strict query comes back empty,
+ to keep the common case at one request.
+ """
+ strict_query = f'artist:"{_escape_lucene(artist)}" AND release:"{_escape_lucene(album)}"'
+ data = await self._get(_BASE_URL + "release", {"query": strict_query})
+ results = (data or {}).get("releases", [])
+ if results:
+ return _best_match_release(artist, album, results)
+
+ loose_query = f"{_escape_lucene(artist)} {_escape_lucene(album)}"
+ data = await self._get(_BASE_URL + "release", {"query": loose_query})
results = (data or {}).get("releases", [])
- return _best_match(album, results, "title")
+ return _best_match_release(artist, album, results)
async def release_details(self, mbid: str) -> dict | None:
"""Full release details, including recordings (tracklist)."""