feedBack/tests/test_mb_match.py
ChrisBeWithYou a65d8cfa13
fix(enrichment): rank the canonical studio take over live/comp versions (#758)
* fix(enrichment): rank the canonical studio take over live/comp versions

A flat MusicBrainz /recording text search ties every take of a song at the
same score, so "AC/DC — Highway to Hell" returns a wall of live bootlegs and
compilations with the 1979 studio version buried (or below the fetch limit).

- build_recording_query: drop live-ONLY recordings (`-secondarytype:Live`).
  Compilations are deliberately kept — they REUSE the studio recording, so
  filtering them cuts the very recording we want (verified against MB).
- _best_release / parse_recording_doc: pick the canonical studio album
  (primary Album, no Live/Compilation/Remix/... secondary type) for the
  displayed album/year, and expose a `studio` flag.
- rank_candidates: since the combined score caps at 1.0 (perfect text match
  ties), break ties on the studio flag and — when the caller knows the audio
  length — on duration proximity, so the studio take wins over live/extended
  cuts. The studio distinction is intentionally NOT scored (a live take is
  still the right SONG), only re-ordered.
- /api/enrichment/search: accept an optional `duration` param so a caller that
  has the audio but no library row (the editor's create modal) can pass the
  master-track length for the duration tiebreak.

Verified end-to-end against live MusicBrainz: AC/DC "Highway to Hell" now
returns the 1979 studio recording at #1 with the correct album + year.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(enrichment): official releases outrank unofficial studio albums

_best_release sorted (clean, status_ok, date), so an UNofficial bootleg Album
outranked an official Single/EP/comp — regressing canonical album/year and
seeding cover-art from a bootleg for single-only songs. Order status_ok before
clean: official first, then prefer a clean studio album among the official
releases (still surfaces the studio album over an official live/comp album).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(enrichment): keep live recordings for genuinely-live charts

build_recording_query unconditionally added -secondarytype:Live, but denoise()
strips a '(Live at …)' qualifier from the query — so a chart that IS a live take
had its only correct recording filtered out (both background enrichment and
manual search). Skip the live filter when the source title carries a
parenthetical live marker; a bare title word ('Live and Let Die') still filters.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(enrichment): drop the studio tiebreak when the chart is a live take

Follow-through on keeping live recordings for live charts: rank_candidates still
ranked the studio take ahead of a tied live one, so a live chart would auto-match
the studio recording. Skip the studio tiebreak when the source title has a live
marker — duration proximity + score then pick the right live version.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-05 00:17:05 +02:00

280 lines
12 KiB
Python

"""Unit tests for lib/mb_match.py — the pure text-matching engine (P8).
No network, no database, no server import: denoise/tokenize, similarity,
scoring + tier classification, Lucene query building, and MusicBrainz
response parsing are all exercised as plain functions.
"""
import mb_match as m
# ── denoise / tokenize ────────────────────────────────────────────────────────
def test_denoise_lowercases_and_strips_punct_and_diacritics():
assert m.denoise("Motörhead") == "motorhead"
assert m.denoise("Beyoncé!!") == "beyonce"
assert m.denoise("Guns N' Roses") == "guns n roses"
assert m.denoise(" Weird spacing ") == "weird spacing"
def test_denoise_strips_noise_parentheticals():
# The design's explicit list: author suffixes + (440Hz)/(Live)/(No Lead)/(v2).
assert m.denoise("Thunderstruck (440Hz)") == "thunderstruck"
assert m.denoise("Thunderstruck (Live)") == "thunderstruck"
assert m.denoise("Thunderstruck (No Lead)") == "thunderstruck"
assert m.denoise("Thunderstruck (v2)") == "thunderstruck"
assert m.denoise("Thunderstruck [Remastered 2012]") == "thunderstruck"
assert m.denoise("One (Live at Wembley)") == "one"
def test_denoise_strips_author_credits():
assert m.denoise("Back in Black (by SomeCharter)") == "back in black"
assert m.denoise("Back in Black (charted by X99)") == "back in black"
assert m.denoise("Back in Black - by SomeCharter") == "back in black"
def test_denoise_keeps_meaningful_parentheticals():
# A parenthetical with no noise term survives (both sides get the same
# treatment, so symmetric content still matches).
assert m.denoise("Doin' It (All for My Baby)") == "doin it all for my baby"
def test_denoise_leading_the_is_artist_only():
assert m.denoise("The Beatles", strip_leading_the=True) == "beatles"
# Titles keep their "The" — never strip it there.
assert m.denoise("The Trooper") == "the trooper"
def test_ampersand_folds_to_and():
assert m.similarity("Angus & Julia Stone", "Angus and Julia Stone", artist=True) == 1.0
# ── similarity ────────────────────────────────────────────────────────────────
def test_similarity_exact_and_empty():
assert m.similarity("Back in Black", "Back In Black!") == 1.0
assert m.similarity("", "Anything") == 0.0
assert m.similarity(None, None) == 0.0
def test_similarity_folds_spelling_drift_via_compaction():
# The headline case: ACDC / AC DC / AC/DC all name the same artist.
assert m.similarity("ACDC", "AC/DC", artist=True) == 1.0
assert m.similarity("AC DC", "ACDC", artist=True) == 1.0
assert m.similarity("Greenday", "Green Day", artist=True) == 1.0
def test_similarity_partial_overlap():
s = m.similarity("Highway to Hell", "Highway Hell")
assert 0.7 < s < 1.0
assert m.similarity("Back in Black", "Paint It Black") < 0.5
# ── scoring + tiers ───────────────────────────────────────────────────────────
SONG = {"artist": "ACDC", "title": "Thunderstruck (v2)", "album": "The Razors Edge",
"year": "1990", "duration": 292}
def test_score_exact_match_is_high():
cand = {"artist": "AC/DC", "title": "Thunderstruck", "year": "1990", "duration": 292}
s = m.score_candidate(SONG, cand)
assert s == 1.0
assert m.classify(SONG, cand, s) == "auto"
def test_score_cover_never_auto():
# Perfect title, wrong artist (a cover) — must not auto-match.
cand = {"artist": "Some Cover Band", "title": "Thunderstruck"}
s = m.score_candidate(SONG, cand)
assert m.classify(SONG, cand, s) != "auto"
def test_missing_artist_caps_at_review():
song = {"artist": "", "title": "Thunderstruck", "duration": 292}
cand = {"artist": "AC/DC", "title": "Thunderstruck", "duration": 292}
s = m.score_candidate(song, cand)
# artist half scores 0 → combined ≤ 0.55 + bonuses → review at best.
assert m.classify(song, cand, s) != "auto"
def test_year_and_duration_corroborate():
# Fuzzy title so the base sits below the 1.0 cap and bonuses are visible.
base = {"artist": "AC/DC", "title": "Thunderstruck Thunder"}
plain = m.score_candidate(SONG, base)
with_year = m.score_candidate(SONG, dict(base, year="1990"))
with_dur = m.score_candidate(SONG, dict(base, duration=290))
assert with_year > plain
assert with_dur > plain
def test_fuzzy_title_with_corroboration_lands_review_or_auto():
cand = {"artist": "AC/DC", "title": "Thunderstruck Thunder"}
s = m.score_candidate(SONG, cand)
assert m.classify(SONG, cand, s) in ("review", "auto")
def test_unrelated_is_none():
cand = {"artist": "Norah Jones", "title": "Sunrise"}
s = m.score_candidate(SONG, cand)
assert m.classify(SONG, cand, s) == "none"
def test_classify_auto_min_override():
# The host's "auto-apply confidence" setting: a perfect match autos at
# any real threshold, and "Always review" (>1.0) sends even it to review.
cand = {"artist": "AC/DC", "title": "Thunderstruck", "year": "1990", "duration": 292}
s = m.score_candidate(SONG, cand)
assert s == 1.0
assert m.classify(SONG, cand, s, auto_min=0.9) == "auto"
assert m.classify(SONG, cand, s, auto_min=1.01) == "review"
# The per-field floors are independent of the threshold: a wrong-artist
# cover stays non-auto even at a permissive auto_min.
cover = {"artist": "Some Cover Band", "title": "Thunderstruck"}
cs = m.score_candidate(SONG, cover)
assert m.classify(SONG, cover, cs, auto_min=0.5) != "auto"
def test_rank_candidates_orders_by_our_score():
cands = [
{"recording_id": "b", "artist": "Someone Else", "title": "Thunderstruck", "mb_score": 100},
{"recording_id": "a", "artist": "AC/DC", "title": "Thunderstruck", "mb_score": 90},
]
ranked = m.rank_candidates(SONG, cands)
assert [c["recording_id"] for c in ranked] == ["a", "b"]
assert all("score" in c for c in ranked)
def test_rank_candidates_studio_preference_is_dropped_for_live_charts():
"""Tied-score candidates: a studio chart prefers the studio take, but a
LIVE chart must NOT be forced to the studio recording."""
studio = {"recording_id": "studio", "artist": "AC/DC", "title": "Highway to Hell",
"studio": True, "mb_score": 90}
live = {"recording_id": "live", "artist": "AC/DC", "title": "Highway to Hell",
"studio": False, "mb_score": 95}
# Studio chart -> studio take wins the tie (studio flag), despite lower mb_score.
studio_song = {"artist": "AC/DC", "title": "Highway to Hell"}
assert m.rank_candidates(studio_song, [live, studio])[0]["recording_id"] == "studio"
# Live chart -> studio preference dropped, so the higher-mb_score live take wins.
live_song = {"artist": "AC/DC", "title": "Highway to Hell (Live at Donington)"}
assert m.rank_candidates(live_song, [studio, live])[0]["recording_id"] == "live"
# ── query building ────────────────────────────────────────────────────────────
def test_build_recording_query_denoises_and_quotes():
q = m.build_recording_query("ACDC", 'Thunderstruck (v2)')
# Live-only recordings are excluded — the studio take is never tagged Live,
# and it's the biggest source of junk in a flat recording search.
assert q == 'recording:"thunderstruck" AND artist:"acdc" AND -secondarytype:Live'
def test_build_recording_query_keeps_live_for_live_charts():
"""A chart that IS a live take must NOT get the live filter, or its only
correct recording is excluded. A bare title word ("Live and Let Die") is a
real word, not a marker, so it still filters."""
live = m.build_recording_query("AC/DC", "Highway to Hell (Live at Donington)")
assert "-secondarytype:Live" not in live
assert 'recording:"highway to hell"' in live
# A real word "live" in the title is not a live marker → still filtered.
bare = m.build_recording_query("Wings", "Live and Let Die")
assert "-secondarytype:Live" in bare
def test_build_recording_query_escapes_and_handles_missing_artist():
q = m.build_recording_query("", 'Say "Hello"')
# Quotes are punct-stripped by denoise, so nothing to escape here — but
# the artist clause must be absent entirely.
assert q.startswith('recording:"')
assert "artist:" not in q
# ── MusicBrainz response parsing ──────────────────────────────────────────────
MB_DOC = {
"id": "rec-123",
"score": 98,
"title": "Thunderstruck",
"length": 292773,
"isrcs": ["AUAP09000045"],
"artist-credit": [
{"name": "AC/DC", "joinphrase": "",
"artist": {"id": "art-1", "name": "AC/DC", "sort-name": "AC/DC"}},
],
"releases": [
{"id": "rel-compilation", "title": "Greatest Hits", "status": "Official",
"date": "2005-01-01", "release-group": {"primary-type": "Compilation"}},
{"id": "rel-album", "title": "The Razors Edge", "status": "Official",
"date": "1990-09-24", "release-group": {"primary-type": "Album"}},
{"id": "rel-boot", "title": "Bootleg", "status": "Bootleg",
"date": "1989-01-01", "release-group": {"primary-type": "Album"}},
],
"tags": [{"name": "hard rock", "count": 10}, {"name": "rock", "count": 4}],
}
def test_parse_recording_doc_normalizes():
c = m.parse_recording_doc(MB_DOC)
assert c["recording_id"] == "rec-123"
assert c["title"] == "Thunderstruck"
assert c["artist"] == "AC/DC"
assert c["artist_id"] == "art-1"
# Official Album beats the compilation and the bootleg.
assert c["album"] == "The Razors Edge"
assert c["release_id"] == "rel-album"
assert c["year"] == "1990"
assert c["duration"] == 293
assert c["isrc"] == "AUAP09000045"
assert c["genres"] == ["hard rock", "rock"]
assert c["mb_score"] == 98
def test_best_release_prefers_official_single_over_unofficial_album():
"""An OFFICIAL single/EP must outrank an UNofficial bootleg album for the
canonical album/year: official comes before the studio-album preference, so
a single-only song is never seeded from a bootleg. (`(clean, status_ok, …)`
would wrongly pick the bootleg.)"""
doc = {
"id": "rec-x", "title": "One-Off", "score": 90,
"artist-credit": [
{"name": "A", "joinphrase": "",
"artist": {"id": "a", "name": "A", "sort-name": "A"}}],
"releases": [
{"id": "rel-boot", "title": "Boot LP", "status": "Bootleg",
"date": "1990-01-01", "release-group": {"primary-type": "Album"}},
{"id": "rel-single", "title": "The Single", "status": "Official",
"date": "1988-01-01", "release-group": {"primary-type": "Single"}},
],
}
c = m.parse_recording_doc(doc)
assert c["release_id"] == "rel-single"
assert c["album"] == "The Single"
assert c["studio"] is False # a Single isn't a clean studio ALBUM
def test_parse_recording_doc_joined_artist_credit():
doc = dict(MB_DOC)
doc["artist-credit"] = [
{"name": "Queen", "joinphrase": " & ",
"artist": {"id": "q", "name": "Queen", "sort-name": "Queen"}},
{"name": "David Bowie",
"artist": {"id": "b", "name": "David Bowie", "sort-name": "Bowie, David"}},
]
c = m.parse_recording_doc(doc)
assert c["artist"] == "Queen & David Bowie"
assert c["artist_id"] == "q"
def test_parse_recording_doc_rejects_malformed():
assert m.parse_recording_doc({}) is None
assert m.parse_recording_doc({"id": "x"}) is None
assert m.parse_recording_doc(None) is None
def test_parse_search_response():
body = {"recordings": [MB_DOC, {"bogus": True}]}
cands = m.parse_search_response(body)
assert len(cands) == 1
assert m.parse_search_response({}) == []
assert m.parse_search_response(None) == []