927266d042
The worker MB resolver picked release groups on (title_ratio, is_album)
and never scored the artist — even though resolve() knows the followed
artist. For a generic title, same-titled release groups by *different*
artists tie on title, and the is_album tiebreak (added in 06e7f91 for
Off the Wall) then actively preferred a foreign Album over the correct
release. Concretely, following Lake Street Dive's "Fun Machine" EP
resolved to an unrelated band "Bastards of Melody"'s same-titled Album,
which then propagated as the downloaded/imported artist.
Make credited-artist similarity the primary sort key, above title and
above is_album. It's a soft signal, never a hard filter, so credited-
name variations (feat., punctuation) still resolve; is_album now only
breaks ties within the same artist+title, preserving the Off the Wall
fix. Verified live: resolve("Lake Street Dive", "Fun Machine") now
returns the 6-track LSD EP.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
121 lines
4.7 KiB
Python
121 lines
4.7 KiB
Python
from difflib import SequenceMatcher
|
|
|
|
from lyra_worker.types import MBTarget
|
|
|
|
_MIN_TITLE_RATIO = 0.6
|
|
|
|
|
|
def _ratio(a: str, b: str) -> float:
|
|
return SequenceMatcher(None, a.strip().casefold(), b.strip().casefold()).ratio()
|
|
|
|
|
|
def _is_album(g: dict) -> bool:
|
|
return (g.get("primary-type") or "").casefold() == "album"
|
|
|
|
|
|
def _credited_artist(g: dict) -> str:
|
|
"""Primary credited artist name of a release-group search result, or '' if absent."""
|
|
credit = g.get("artist-credit")
|
|
if isinstance(credit, list) and credit and isinstance(credit[0], dict):
|
|
return (credit[0].get("artist") or {}).get("name", "") or ""
|
|
return ""
|
|
|
|
|
|
def _best_release_group(artist: str, album: str, groups: list) -> dict | None:
|
|
"""Choose the best release-group match for `artist`/`album`.
|
|
|
|
Ranked, highest first, on `(artist_ratio, title_ratio, is_album)`:
|
|
* artist match dominates — MusicBrainz relevance ties same-titled release groups by
|
|
*different* artists (e.g. "Fun Machine" exists as a Lake Street Dive EP and an
|
|
unrelated band's Album), and the artist we followed is trustworthy, so a credited
|
|
artist that matches the request outranks everything else;
|
|
* title similarity is next;
|
|
* an Album primary-type only breaks ties *within* the same artist+title, so a famous
|
|
album (Michael Jackson's "Off the Wall") still isn't resolved to its same-named
|
|
single whose short tracklist would map positionally onto the album's files.
|
|
Artist is a ranking signal, never a hard filter — a low match sinks a candidate but
|
|
never drops the release, so credited-name variations (feat., punctuation) still resolve.
|
|
Returns None below the title threshold. Pure — no network I/O."""
|
|
if not groups:
|
|
return None
|
|
best = max(
|
|
groups,
|
|
key=lambda g: (
|
|
_ratio(artist, _credited_artist(g)),
|
|
_ratio(album, g.get("title", "")),
|
|
_is_album(g),
|
|
),
|
|
)
|
|
if _ratio(album, best.get("title", "")) < _MIN_TITLE_RATIO:
|
|
return None
|
|
return best
|
|
|
|
|
|
class MusicBrainzResolver:
|
|
"""Real MbResolver using the MusicBrainz webservice via musicbrainzngs.
|
|
|
|
NOT unit-tested offline; see test_musicbrainz_live.py. musicbrainzngs is
|
|
imported lazily so importing/constructing this class stays offline.
|
|
"""
|
|
|
|
def __init__(self, app_name: str = "Lyra", version: str = "0.1", contact: str = "lyra@localhost"):
|
|
self._app = app_name
|
|
self._version = version
|
|
self._contact = contact
|
|
|
|
def resolve(self, artist: str, album: str) -> MBTarget | None:
|
|
import musicbrainzngs
|
|
|
|
musicbrainzngs.set_useragent(self._app, self._version, self._contact)
|
|
|
|
res = musicbrainzngs.search_release_groups(query=album, artist=artist, limit=5)
|
|
groups = res.get("release-group-list", [])
|
|
rg = _best_release_group(artist, album, groups)
|
|
if rg is None:
|
|
return None
|
|
|
|
canonical_album = rg.get("title", album)
|
|
canonical_artist = artist
|
|
artist_mbid = ""
|
|
credit = rg.get("artist-credit")
|
|
if isinstance(credit, list) and credit and isinstance(credit[0], dict):
|
|
_artist = credit[0].get("artist") or {}
|
|
canonical_artist = _artist.get("name", artist)
|
|
artist_mbid = _artist.get("id", "") or ""
|
|
|
|
year = None
|
|
frd = rg.get("first-release-date", "") or ""
|
|
if frd[:4].isdigit():
|
|
year = int(frd[:4])
|
|
|
|
track_count = None
|
|
total_duration_s = None
|
|
titles: tuple[str, ...] = ()
|
|
rgid = rg.get("id")
|
|
if rgid:
|
|
rgfull = musicbrainzngs.get_release_group_by_id(rgid, includes=["releases"])
|
|
releases = rgfull.get("release-group", {}).get("release-list", [])
|
|
if releases:
|
|
rel = musicbrainzngs.get_release_by_id(
|
|
releases[0]["id"], includes=["recordings"]
|
|
).get("release", {})
|
|
tracks = []
|
|
for medium in rel.get("medium-list", []):
|
|
tracks.extend(medium.get("track-list", []))
|
|
if tracks:
|
|
track_count = len(tracks)
|
|
total_ms = sum(int((t.get("recording") or {}).get("length") or 0) for t in tracks)
|
|
total_duration_s = total_ms // 1000 if total_ms else None
|
|
titles = tuple((t.get("recording") or {}).get("title", "") for t in tracks)
|
|
|
|
return MBTarget(
|
|
artist=canonical_artist,
|
|
album=canonical_album,
|
|
track_count=track_count,
|
|
total_duration_s=total_duration_s,
|
|
year=year,
|
|
tracklist=titles,
|
|
rg_mbid=rg.get("id", "") or "",
|
|
artist_mbid=artist_mbid,
|
|
)
|