Files
Lyra/worker/lyra_worker/_musicbrainz.py
T
Jonathan 927266d042 fix(mb): rank release groups by credited artist, not title alone
The worker MB resolver picked release groups on (title_ratio, is_album)
and never scored the artist — even though resolve() knows the followed
artist. For a generic title, same-titled release groups by *different*
artists tie on title, and the is_album tiebreak (added in 06e7f91 for
Off the Wall) then actively preferred a foreign Album over the correct
release. Concretely, following Lake Street Dive's "Fun Machine" EP
resolved to an unrelated band "Bastards of Melody"'s same-titled Album,
which then propagated as the downloaded/imported artist.

Make credited-artist similarity the primary sort key, above title and
above is_album. It's a soft signal, never a hard filter, so credited-
name variations (feat., punctuation) still resolve; is_album now only
breaks ties within the same artist+title, preserving the Off the Wall
fix. Verified live: resolve("Lake Street Dive", "Fun Machine") now
returns the 6-track LSD EP.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-15 12:32:12 +02:00

121 lines
4.7 KiB
Python

from difflib import SequenceMatcher
from lyra_worker.types import MBTarget
_MIN_TITLE_RATIO = 0.6
def _ratio(a: str, b: str) -> float:
return SequenceMatcher(None, a.strip().casefold(), b.strip().casefold()).ratio()
def _is_album(g: dict) -> bool:
return (g.get("primary-type") or "").casefold() == "album"
def _credited_artist(g: dict) -> str:
"""Primary credited artist name of a release-group search result, or '' if absent."""
credit = g.get("artist-credit")
if isinstance(credit, list) and credit and isinstance(credit[0], dict):
return (credit[0].get("artist") or {}).get("name", "") or ""
return ""
def _best_release_group(artist: str, album: str, groups: list) -> dict | None:
"""Choose the best release-group match for `artist`/`album`.
Ranked, highest first, on `(artist_ratio, title_ratio, is_album)`:
* artist match dominates — MusicBrainz relevance ties same-titled release groups by
*different* artists (e.g. "Fun Machine" exists as a Lake Street Dive EP and an
unrelated band's Album), and the artist we followed is trustworthy, so a credited
artist that matches the request outranks everything else;
* title similarity is next;
* an Album primary-type only breaks ties *within* the same artist+title, so a famous
album (Michael Jackson's "Off the Wall") still isn't resolved to its same-named
single whose short tracklist would map positionally onto the album's files.
Artist is a ranking signal, never a hard filter — a low match sinks a candidate but
never drops the release, so credited-name variations (feat., punctuation) still resolve.
Returns None below the title threshold. Pure — no network I/O."""
if not groups:
return None
best = max(
groups,
key=lambda g: (
_ratio(artist, _credited_artist(g)),
_ratio(album, g.get("title", "")),
_is_album(g),
),
)
if _ratio(album, best.get("title", "")) < _MIN_TITLE_RATIO:
return None
return best
class MusicBrainzResolver:
"""Real MbResolver using the MusicBrainz webservice via musicbrainzngs.
NOT unit-tested offline; see test_musicbrainz_live.py. musicbrainzngs is
imported lazily so importing/constructing this class stays offline.
"""
def __init__(self, app_name: str = "Lyra", version: str = "0.1", contact: str = "lyra@localhost"):
self._app = app_name
self._version = version
self._contact = contact
def resolve(self, artist: str, album: str) -> MBTarget | None:
import musicbrainzngs
musicbrainzngs.set_useragent(self._app, self._version, self._contact)
res = musicbrainzngs.search_release_groups(query=album, artist=artist, limit=5)
groups = res.get("release-group-list", [])
rg = _best_release_group(artist, album, groups)
if rg is None:
return None
canonical_album = rg.get("title", album)
canonical_artist = artist
artist_mbid = ""
credit = rg.get("artist-credit")
if isinstance(credit, list) and credit and isinstance(credit[0], dict):
_artist = credit[0].get("artist") or {}
canonical_artist = _artist.get("name", artist)
artist_mbid = _artist.get("id", "") or ""
year = None
frd = rg.get("first-release-date", "") or ""
if frd[:4].isdigit():
year = int(frd[:4])
track_count = None
total_duration_s = None
titles: tuple[str, ...] = ()
rgid = rg.get("id")
if rgid:
rgfull = musicbrainzngs.get_release_group_by_id(rgid, includes=["releases"])
releases = rgfull.get("release-group", {}).get("release-list", [])
if releases:
rel = musicbrainzngs.get_release_by_id(
releases[0]["id"], includes=["recordings"]
).get("release", {})
tracks = []
for medium in rel.get("medium-list", []):
tracks.extend(medium.get("track-list", []))
if tracks:
track_count = len(tracks)
total_ms = sum(int((t.get("recording") or {}).get("length") or 0) for t in tracks)
total_duration_s = total_ms // 1000 if total_ms else None
titles = tuple((t.get("recording") or {}).get("title", "") for t in tracks)
return MBTarget(
artist=canonical_artist,
album=canonical_album,
track_count=track_count,
total_duration_s=total_duration_s,
year=year,
tracklist=titles,
rg_mbid=rg.get("id", "") or "",
artist_mbid=artist_mbid,
)