mirror of
https://github.com/got-feedBack/feedBack.git
synced 2026-09-15 17:30:13 +00:00
feat(enrichment): AcoustID audio-fingerprint identification (opt-in)
Text search can only guess the version; the definitive fix is content-based — fingerprint the actual audio with Chromaprint (fpcalc) and look it up on AcoustID, which maps the fingerprint to the EXACT MusicBrainz recording (the approach Lidarr uses). Sidesteps the studio-vs-live ambiguity entirely. - lib/acoustid_match.py: pure response parsing + config gating (unit-tested); normalizes AcoustID hits into the same candidate shape as mb_match so the review UI + editor Match popup render fingerprint and text hits identically. - server.py: _fpcalc (Chromaprint subprocess), _acoustid_lookup (throttled, offline-guarded HTTP), _identify_by_fingerprint (also available to the library-enrichment pipeline), and POST /api/enrichment/identify (upload the master audio → candidates). - Fully OPT-IN and graceful: absent the fpcalc binary or an ACOUSTID_API_KEY the whole path is a no-op / 503 and the text matcher runs unchanged. Requires (both optional): the `fpcalc` (Chromaprint) binary on PATH/$FPCALC, and a free AcoustID application key in $ACOUSTID_API_KEY. Pure parsing/gating is unit-tested; the fpcalc + live-lookup path needs those two to exercise. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
b6169af6aa
commit
0bdf0f1311
@@ -0,0 +1,94 @@
|
||||
"""Pure-function tests for AcoustID fingerprint response parsing + config
|
||||
gating. No network, no fpcalc binary — server.py owns those seams."""
|
||||
import acoustid_match as a
|
||||
|
||||
|
||||
def _resp(score=0.97, rec_id="rec-1", title="Highway to Hell", artist="AC/DC",
|
||||
rg_title="Highway to Hell", rg_type="Album", secondary=None,
|
||||
year=1979, duration=208.4):
|
||||
return {
|
||||
"status": "ok",
|
||||
"results": [{
|
||||
"id": "acoustid-uuid",
|
||||
"score": score,
|
||||
"recordings": [{
|
||||
"id": rec_id,
|
||||
"title": title,
|
||||
"duration": duration,
|
||||
"artists": [{"id": "a1", "name": artist}],
|
||||
"releasegroups": [{
|
||||
"id": "rg1", "title": rg_title, "type": rg_type,
|
||||
"secondarytypes": secondary or [],
|
||||
"releases": [{"date": {"year": year}}],
|
||||
}],
|
||||
}],
|
||||
}],
|
||||
}
|
||||
|
||||
|
||||
def test_parse_maps_the_studio_recording():
|
||||
out = a.parse_lookup_response(_resp())
|
||||
assert len(out) == 1
|
||||
c = out[0]
|
||||
assert c["recording_id"] == "rec-1"
|
||||
assert c["title"] == "Highway to Hell"
|
||||
assert c["artist"] == "AC/DC"
|
||||
assert c["album"] == "Highway to Hell"
|
||||
assert c["year"] == "1979"
|
||||
assert c["duration"] == 208
|
||||
assert c["studio"] is True
|
||||
assert c["source"] == "acoustid"
|
||||
assert c["mb_score"] == 97 # 0.97 → 0..100 confidence band
|
||||
assert c["score"] == 0.97
|
||||
|
||||
|
||||
def test_live_release_group_is_not_studio():
|
||||
out = a.parse_lookup_response(_resp(rg_type="Album", secondary=["Live"]))
|
||||
assert out[0]["studio"] is False
|
||||
|
||||
|
||||
def test_compilation_is_not_studio():
|
||||
out = a.parse_lookup_response(_resp(secondary=["Compilation"]))
|
||||
assert out[0]["studio"] is False
|
||||
|
||||
|
||||
def test_prefers_studio_group_for_album_display():
|
||||
resp = _resp()
|
||||
# Add a comp release-group first; the studio one must win the album pick.
|
||||
resp["results"][0]["recordings"][0]["releasegroups"].insert(0, {
|
||||
"id": "rg0", "title": "Greatest Hits", "type": "Album",
|
||||
"secondarytypes": ["Compilation"], "releases": [{"date": {"year": 2000}}],
|
||||
})
|
||||
c = a.parse_lookup_response(resp)[0]
|
||||
assert c["album"] == "Highway to Hell"
|
||||
assert c["studio"] is True
|
||||
|
||||
|
||||
def test_dedupes_recording_across_results():
|
||||
resp = _resp()
|
||||
resp["results"].append(dict(resp["results"][0])) # same recording again
|
||||
assert len(a.parse_lookup_response(resp)) == 1
|
||||
|
||||
|
||||
def test_non_ok_status_and_garbage_return_empty():
|
||||
assert a.parse_lookup_response({"status": "error"}) == []
|
||||
assert a.parse_lookup_response({}) == []
|
||||
assert a.parse_lookup_response(None) == []
|
||||
assert a.parse_lookup_response({"status": "ok", "results": []}) == []
|
||||
|
||||
|
||||
def test_higher_acoustid_score_ranks_first():
|
||||
resp = _resp(score=0.55, rec_id="low")
|
||||
resp["results"].append(_resp(score=0.99, rec_id="high")["results"][0])
|
||||
out = a.parse_lookup_response(resp)
|
||||
assert out[0]["recording_id"] == "high"
|
||||
|
||||
|
||||
def test_config_gating(monkeypatch):
|
||||
monkeypatch.delenv("ACOUSTID_API_KEY", raising=False)
|
||||
assert a.api_key() == ""
|
||||
assert a.is_configured() is False
|
||||
assert a.is_configured("explicit-key") is True
|
||||
monkeypatch.setenv("ACOUSTID_API_KEY", "envkey")
|
||||
assert a.api_key() == "envkey"
|
||||
assert a.is_configured() is True
|
||||
Reference in New Issue
Block a user