From c797cf570c639a3917633e560ac4fe4e88bfb058 Mon Sep 17 00:00:00 2001 From: trav Date: Fri, 11 Sep 2026 20:41:33 -0500 Subject: [PATCH] v0.20.1: one fingerprint, six artists MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit trav's "Dionne Farris - I Know" kept identifying as Jay-Z, against ID3 tags that plainly said otherwise. AcoustID was right; the ranking threw the answer away. The fingerprint matched one AcoustID result at 0.97, and six recordings hang off it: Dionne Farris twice, plus Jay-Z, Marisela, New Atlantic and David Essex, all of whom recorded a song called "I Know". A result's score belongs to the *audio*, so every linked recording carries it however wrong the link is. With the scores tied, ranking fell through to the duration bucket, where Jay-Z's 222.7 s beat Dionne's 227.3 s against a 224 s file. The tags never got a vote: the hint sat below duration in the sort key. So the lookup now asks who submitted each link. `sources` joins LOOKUP_META — 475 people linked that audio to Dionne Farris, 6 to Jay-Z, 1 each to the rest — and _link_tier sinks anything under a tenth of the strongest link in the same result. The share is relative, never an absolute count, and a missing count ranks as real: an obscure song's true link may have two submissions against a stray's one, and rounds 45-46's payloads rank unchanged. And it asks what the file already says. artist_hint_for gathers the artist tag, the album artist and the artist in the filename; _artist_agreement counts the words shared with a candidate's credit, placeholders dropped. Like every hint since round 46 it only chooses among what AcoustID returned. New key order: stray tier, artist agreement, duration bucket, hint overlap, release rank — who, which take, which release. Artist above duration is the whole fix; duration still separates two takes by one artist. Verified live against the reported file: the proposal is now I Know — Dionne Farris — Wild Seed - Wild Flower (1994), track 1, with all five mis-tagged artists off the dropdown. The real response is pinned in tests/test_round52.py. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01SKXUgsBBwe3qaHEjeV8ubP --- CLAUDE.md | 14 ++- lintunes/__init__.py | 2 +- lintunes/fingerprint.py | 110 +++++++++++++++++-- tasks-done.md | 43 ++++++++ tests/test_round52.py | 227 ++++++++++++++++++++++++++++++++++++++++ 5 files changed, 383 insertions(+), 13 deletions(-) create mode 100644 tests/test_round52.py diff --git a/CLAUDE.md b/CLAUDE.md index 78da055..69325b0 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -207,7 +207,19 @@ persistence) → GUI (Qt widgets that read the manager and connect to its signal compilation as the album, because a 60s song must not sort by the year its CD reissue came out. `_releasegroup_sort_key` separately ranks a plain studio Album above EP/Single above anything with a Compilation/Live secondary type, - so the default proposal is the real album. `gui/identify_dialog.py` is passive + so the default proposal is the real album. **An AcoustID result's score + belongs to the *audio*, not to any one recording**: every recording linked to + it carries that score, mis-tags included, and a song whose title another + artist also used collects them (Round 52: trav's Dionne Farris "I Know" is + linked to Jay-Z's, Marisela's, David Essex's and New Atlantic's). What tells + a real link from a stray is `sources`, the submission count — 475 against 6 + and three 1s — so the lookup asks for it and `_link_tier` sinks anything + under a tenth of the strongest link in that result. Hence the key order + *who*, then *which take*, then *which release*: stray tier, agreement with + the artist the file's own tags name (`artist_hint_for`), duration bucket, + hint overlap, release rank. Artist before duration is the point — Jay-Z's + take was 1.5 s closer to the file than Dionne's own, and that used to decide + it. `gui/identify_dialog.py` is passive (the `AlbumArtDialog` contract — nothing written until accepted); the caller applies `result_fields()` through `LibraryManager.edit_track_fields`, which is what buys tag writes, undo and artist/album file relocation for free. A row is diff --git a/lintunes/__init__.py b/lintunes/__init__.py index 1693ef7..0d87d85 100644 --- a/lintunes/__init__.py +++ b/lintunes/__init__.py @@ -1,3 +1,3 @@ """LinTunes — iTunes-style music library manager and player for Linux.""" -__version__ = "0.20.0" +__version__ = "0.20.1" diff --git a/lintunes/fingerprint.py b/lintunes/fingerprint.py index ebff5c3..ef1316f 100644 --- a/lintunes/fingerprint.py +++ b/lintunes/fingerprint.py @@ -25,7 +25,7 @@ from lintunes import filename_tags ACOUSTID_LOOKUP_URL = "https://api.acoustid.org/v2/lookup" -LOOKUP_META = "recordings releasegroups releases tracks compress" +LOOKUP_META = "recordings releasegroups releases tracks sources compress" SCORE_THRESHOLD = 0.5 # AcoustID scores below this are noise MAX_CANDIDATES = 8 TIMEOUT_S = 15 @@ -155,6 +155,63 @@ def _hint_overlap(hint_tokens: set, candidate: IdentifyCandidate) -> int: for token in hint_tokens & _tokens(text)) +# A link backed by fewer than this share of the strongest link's submissions +# is a stray rather than a rival: 6 against 475 is somebody's mis-tag, while +# 3 against 20 is a genuinely contested song. +STRAY_LINK_SHARE = 0.1 + + +def _link_tier(recording: dict, strongest: int) -> int: + """0 for a real link, 1 for a stray, by how many people submitted it. + + One AcoustID result is one piece of *audio*, and its score says how well + the file matched that audio — so every recording linked to it shares that + score, however wrong the link is. Links are user-submitted, and a song + whose title another artist also used collects mis-tags: trav's Dionne + Farris "I Know" is linked to Jay-Z's "I Know", Marisela's, David Essex's + and New Atlantic's. What separates them is how many people submitted each + link — 475 against 6, 1, 1 and 1 — which is the only field in the response + that knows the difference. + + Relative to the strongest link, never an absolute count: an obscure song's + real link may have two submissions against a stray's one. Unknown counts + (an older payload, or a result where nobody reported any) rank as real, so + nothing is demoted on missing evidence. + """ + sources = recording.get("sources") + if not isinstance(sources, int) or strongest <= 0: + return 0 + return 1 if sources < strongest * STRAY_LINK_SHARE else 0 + + +def _strongest_link(result: dict) -> int: + """The most submissions behind any one link of an AcoustID result.""" + counts = [r.get("sources") for r in result.get("recordings", [])] + return max((c for c in counts if isinstance(c, int)), default=0) + + +# Words that name no artist, so agreeing with them means nothing. +_PLACEHOLDER_ARTIST = {"unknown", "various", "artist", "artists"} + + +def _artist_tokens(text: str) -> set: + """Comparable words from an artist credit, placeholders dropped.""" + return {t for t in _tokens(text) if not t.isdigit()} - _PLACEHOLDER_ARTIST + + +def _artist_agreement(artist_tokens: set, candidate: IdentifyCandidate) -> int: + """How much a candidate's artist echoes the one the file already names. + + Only ever chooses *among* the recordings AcoustID linked to this audio, so + like the rest of the hint it can never invent a candidate — it just stops a + file whose tags plainly say "dionne farris" from being told it is Jay-Z. + """ + if not artist_tokens: + return 0 + text = " ".join(p for p in (candidate.artist, candidate.album_artist) if p) + return len(artist_tokens & _artist_tokens(text)) + + def _pick_release(releasegroup: dict) -> dict | None: """Earliest dated release that carries mediums (track positions); falls back to the earliest dated one, then the first.""" @@ -192,7 +249,8 @@ def _track_position(release: dict | None) -> tuple[int | None, int | None, def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, limit: int = MAX_CANDIDATES, hint: str = "", - duration: float | None = None) -> list[IdentifyCandidate]: + duration: float | None = None, + artist_hint: str = "") -> list[IdentifyCandidate]: """Turn an AcoustID lookup response into ranked candidates. One candidate per release group of each matched recording, so the @@ -203,20 +261,32 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, came out, not by which pressing this file happens to be from. ``hint`` is whatever the file already claims to be (its tags and its - filename). It never invents a candidate, but it breaks ties: a release - whose words echo the hint outranks one that doesn't, and a year named in - the hint wins when it predates anything the database knows — MusicBrainz - often holds only a reissue's date, with the original pressing undated. + filename), and ``artist_hint`` is just the artist part of that. Neither + invents a candidate; they decide between the ones AcoustID returned. + + The order is *who*, then *which take*, then *which release*: a stray link + (see ``_link_tier``) sinks below the links people actually submitted, then + an artist agreeing with the file's own tags wins, then the recording whose + length matches, and only then the release whose words echo the hint. Artist + before duration is the Dionne Farris rule — five artists' mis-tags hang off + that one fingerprint, and Jay-Z's "I Know" happened to be 1.5 s closer. + + A year named in the hint still wins when it predates anything the database + knows: MusicBrainz often holds only a reissue's date, with the original + pressing undated. """ hint_tokens = _tokens(hint) hint_year = _single_year(hint) + artist_tokens = _artist_tokens(artist_hint) scored = [] seen = set() for result in payload.get("results", []): score = result.get("score", 0) if score < threshold: continue + strongest = _strongest_link(result) for recording in result.get("recordings", []): + tier = _link_tier(recording, strongest) length = _duration_bucket(recording, duration) title = recording.get("title") artist = _join_artists(recording.get("artists")) @@ -232,7 +302,9 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, if title: bare = IdentifyCandidate(score=score, name=title, artist=artist, year=hint_year) - scored.append(((-round(score, 2), length, + scored.append(((-round(score, 2), tier, + -_artist_agreement(artist_tokens, bare), + length, -_hint_overlap(hint_tokens, bare), 4, 9999), bare)) continue @@ -256,7 +328,9 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, disc_number=disc, ) rank, rg_year = _releasegroup_sort_key(releasegroup) - scored.append(((-round(score, 2), length, + scored.append(((-round(score, 2), tier, + -_artist_agreement(artist_tokens, candidate), + length, -_hint_overlap(hint_tokens, candidate), rank, rg_year), candidate)) scored.sort(key=lambda pair: pair[0]) @@ -278,6 +352,16 @@ def hint_for(track) -> str: return " ".join(p for p in parts if p) +def artist_hint_for(track) -> str: + """Who the file already says made this — its artist tags plus the artist + its own name carries. Ranks the recordings linked to one fingerprint.""" + parts = [track.artist or "", getattr(track, "album_artist", "") or ""] + if track.location: + parts.append(filename_tags.parse_filename(track.location) + .get("artist", "")) + return " ".join(p for p in parts if p) + + def candidate_from_filename(track) -> IdentifyCandidate | None: """A proposal read off the file's own name, for the very common case that no database has ever heard of the song. @@ -328,7 +412,8 @@ def fingerprint_file(location: str) -> tuple[int, str]: def lookup_fingerprint(api_key: str, duration: int, fingerprint: str, - hint: str = "") -> list[IdentifyCandidate]: + hint: str = "", + artist_hint: str = "") -> list[IdentifyCandidate]: import requests response = requests.post( ACOUSTID_LOOKUP_URL, @@ -342,7 +427,8 @@ def lookup_fingerprint(api_key: str, duration: int, fingerprint: str, if payload.get("status") != "ok": message = (payload.get("error") or {}).get("message", "lookup failed") raise RuntimeError(message) - return parse_lookup(payload, hint=hint, duration=duration) + return parse_lookup(payload, hint=hint, duration=duration, + artist_hint=artist_hint) class TrackIdentifier(QObject): @@ -360,6 +446,7 @@ class TrackIdentifier(QObject): track_id = track.track_id location = track.location hint = hint_for(track) + artist_hint = artist_hint_for(track) guess = candidate_from_filename(track) def work(): @@ -372,7 +459,8 @@ class TrackIdentifier(QObject): if not api_key: raise RuntimeError("no AcoustID key set") duration, fp = fingerprint_file(location) - candidates = lookup_fingerprint(api_key, duration, fp, hint) + candidates = lookup_fingerprint(api_key, duration, fp, hint, + artist_hint) except Exception as e: # A file too short or too odd to fingerprint still has a name. if guess is not None: diff --git a/tasks-done.md b/tasks-done.md index 19502a3..955c7b7 100644 --- a/tasks-done.md +++ b/tasks-done.md @@ -1,5 +1,48 @@ ## Done +### Round 52 (2026-09-11) — one fingerprint, six artists (v0.20.1) + +trav's "Dionne Farris - I Know" kept being identified as Jay-Z, with ID3 tags +that plainly said otherwise. AcoustID was right all along; the ranking threw +the answer away. + +- [x] **The diagnosis.** The fingerprint matched one AcoustID result at 0.97, + and *six* recordings hang off it — Dionne Farris twice, plus Jay-Z, + Marisela, New Atlantic and David Essex, all of whom recorded a song + called "I Know". A result's score belongs to the **audio**, so every + linked recording carries it however wrong the link is. With the scores + tied, ranking fell through to the duration bucket, and Jay-Z's 222.7 s + beat Dionne's 227.3 s against a 224 s file. The tags were never consulted + — the hint sat *below* duration in the sort key. +- [x] **Ask who submitted the link.** `sources` joins `LOOKUP_META`: 475 people + linked that audio to Dionne Farris, 6 to Jay-Z, and 1 each to the other + three. `_link_tier` sinks anything under a tenth of the strongest link in + the same result. The share is **relative**, never an absolute count, and + a missing count ranks as real — an obscure song's true link may have two + submissions against a stray's one, and nothing is demoted on missing + evidence (every canned payload from rounds 45–46 ranks unchanged). +- [x] **Ask what the file already says.** `artist_hint_for` gathers the artist + tag, the album artist and the artist in the filename; `_artist_agreement` + counts the words it shares with a candidate's artist credit, with + placeholders ("Unknown Artist", "Various") dropped. Like every hint since + round 46 it only chooses *among* what AcoustID returned — it can't invent + a candidate. +- [x] **New key order: who, which take, which release** — stray tier, artist + agreement, duration bucket, hint overlap, release rank. Artist above + duration is the whole fix; duration still separates two takes by one + artist, so round 46's rule is intact. +- [x] Verified live against the reported file: the proposal is now + **I Know — Dionne Farris — Wild Seed - Wild Flower (1994)**, track 1, and + all five mis-tagged artists are off the eight-candidate dropdown. The + real response is pinned as a canned payload in `tests/test_round52.py`. +- [x] **Noticed while verifying**: the full suite took 550 s and + `test_round33.py::…::test_stops_when_the_playing_track_is_deleted` hung + outright, with the R1 mounted. Not the Round 51 wedge — no uninterruptible + FUSE waits, and it cleared on its own: a later run was 932 passed in + 12.2 s with that test green, and it passes 5/5 in 0.19 s alone. So an + MTP mount that is merely *busy* (not wedged) slows the suite ~45× and can + stall a Qt test outright. Worth suspecting before believing a hang. + ### Round 51 (2026-09-11) — counts come home, and only playable files go (v0.20.0) andTunes phase 3. The Rabbit is now one more machine in the play-count diff --git a/tests/test_round52.py b/tests/test_round52.py new file mode 100644 index 0000000..80d7bab --- /dev/null +++ b/tests/test_round52.py @@ -0,0 +1,227 @@ +"""Round 52 — Identify Track stops calling Dionne Farris "Jay-Z". + +trav's "Dionne Farris - I Know [fqng9NDqKB8].mp3" kept being identified as +Jay-Z's "I Know" from American Gangster, and AcoustID was not at fault. The +fingerprint matched one AcoustID result at 0.97, and *six* recordings hang off +that one result — Dionne Farris twice, plus Jay-Z, Marisela, New Atlantic and +David Essex, all of whom recorded a song called "I Know". The score belongs to +the audio, so all six carried 0.97, and ranking fell through to the duration +bucket, where Jay-Z's 222.7 s beat Dionne's 227.3 s against a 224 s file. The +tags said "dionne farris" the whole time. + +Two things were missing. The lookup never asked for ``sources`` — how many +people submitted each link, which is 475 for Dionne against 6 for Jay-Z and 1 +for the rest — and nothing compared the candidates' artist against the file's +own. So the ranking now goes: strays below real links, then the artist the file +names, then the take whose length matches, then the release the hint echoes. +""" +import pytest + +from lintunes import fingerprint +from lintunes.fingerprint import ( + artist_hint_for, lookup_fingerprint, parse_lookup, +) + + +class _Track: + def __init__(self, name="", artist="", album="", album_artist="", + location=""): + self.track_id = 1 + self.name = name + self.artist = artist + self.album = album + self.album_artist = album_artist + self.location = location + + +def _group(title, year, type_="Album", secondary=None): + return {"title": title, "type": type_, + "secondarytypes": list(secondary or []), + "releases": [{"date": {"year": year} if year else {}}]} + + +def _recording(title, artist, duration=None, sources=None, groups=()): + rec = {"title": title, "artists": [{"name": artist}], + "releasegroups": list(groups)} + if duration is not None: + rec["duration"] = duration + if sources is not None: + rec["sources"] = sources + return rec + + +def _payload(*recordings, score=0.9695): + return {"status": "ok", + "results": [{"score": score, "recordings": list(recordings)}]} + + +# The real response, trimmed to the release groups that matter. Durations and +# source counts are exactly what AcoustID returned for trav's file; the file +# itself is 224 s, which is why Jay-Z used to win. +def _i_know_payload(): + return _payload( + _recording("I Know", "Marisela", duration=227.8, sources=1, + groups=[_group("Salsa: Original Motion Picture Soundtrack", + 1988)]), + _recording("I Know (NY reprise mix)", "Dionne Farris", duration=229.4, + sources=11, groups=[_group("I Know", 1994, type_="Single")]), + _recording("I Know", "Jay‐Z", duration=222.706, sources=6, + groups=[_group("American Gangster", 2007)]), + _recording("I Know", "New Atlantic", duration=238.986, sources=1, + groups=[_group("Very Best of Back to the Old Skool", 2001, + secondary=["Compilation"])]), + _recording("I Know", "David Essex", duration=214.0, sources=1, + groups=[_group("David Essex", 1974)]), + _recording("I Know", "Dionne Farris", duration=227.266, sources=475, + groups=[_group("Wild Seed - Wild Flower", 1994), + _group("The Hangover Cure: Time to Chill", 1996, + secondary=["Compilation"])]), + ) + + +DIONNE_TRACK = _Track( + name="I Know", artist="dionne farris", album="Wild Seed – Wild Flower", + location="/m/dionne farris/Wild Seed – Wild Flower/" + "Dionne Farris - I Know [fqng9NDqKB8].mp3") + + +class TestTheDionneFarrisCase: + def test_the_file_is_no_longer_told_it_is_jay_z(self): + top = parse_lookup(_i_know_payload(), duration=224, + hint=fingerprint.hint_for(DIONNE_TRACK), + artist_hint=artist_hint_for(DIONNE_TRACK))[0] + assert top.artist == "Dionne Farris" + assert top.album == "Wild Seed - Wild Flower" + assert top.year == 1994 + + def test_submissions_alone_settle_it_when_the_file_says_nothing(self): + """An untagged rip has no artist to agree with, and 475 links against + 6 still say whose song this is.""" + top = parse_lookup(_i_know_payload(), duration=224)[0] + assert top.artist == "Dionne Farris" + + def test_a_stray_link_cannot_confirm_a_wrong_tag(self): + """Tagged Jay-Z, so agreement points at the mis-tag — but 6 links + against 475 is a stray, and strays rank below real links.""" + wrong = _Track(name="I Know", artist="Jay-Z", location="/m/x/y/z.mp3") + top = parse_lookup(_i_know_payload(), duration=224, + hint=fingerprint.hint_for(wrong), + artist_hint=artist_hint_for(wrong))[0] + assert top.artist == "Dionne Farris" + + def test_a_closer_duration_no_longer_outranks_the_artist(self): + """The whole bug in one line: Jay-Z's take is the closest match by + length, and it still must not win.""" + best_length = min( + _i_know_payload()["results"][0]["recordings"], + key=lambda r: abs(r["duration"] - 224)) + assert best_length["artists"][0]["name"] == "Jay‐Z" + top = parse_lookup(_i_know_payload(), duration=224, + artist_hint="Dionne Farris")[0] + assert top.artist == "Dionne Farris" + + +class TestContestedLinks: + """Two links people genuinely submitted, rather than one and a stray.""" + + def _contested(self): + return _payload( + _recording("Shadow", "The Loud Ones", duration=180.0, sources=20, + groups=[_group("Loud Album", 2001)]), + _recording("Shadow", "Quiet Hour", duration=180.0, sources=3, + groups=[_group("Quiet Album", 2001)]), + ) + + def test_the_files_own_tags_decide(self): + top = parse_lookup(self._contested(), duration=180, + artist_hint="Quiet Hour")[0] + assert top.artist == "Quiet Hour" + + def test_without_tags_the_stronger_link_leads(self): + assert parse_lookup(self._contested(), duration=180)[0].artist == \ + "The Loud Ones" + + def test_a_tenth_is_the_line_between_a_rival_and_a_stray(self): + assert fingerprint._link_tier({"sources": 3}, 20) == 0 + assert fingerprint._link_tier({"sources": 1}, 20) == 1 + + +class TestNoEvidenceChangesNothing: + def test_a_payload_without_sources_ranks_as_before(self): + """Older responses (and every canned payload in rounds 45–46) carry no + counts, so nothing may be demoted for lacking them.""" + payload = _payload( + _recording("Snowfall", "The Band", duration=155.0, + groups=[_group("The Long Take", 2000)]), + _recording("Snowfall", "The Band", duration=150.3, + groups=[_group("The Short Take", 2000)]), + ) + assert parse_lookup(payload, duration=150)[0].album == "The Short Take" + + def test_a_result_where_nobody_reported_counts_demotes_nothing(self): + assert fingerprint._link_tier({}, 0) == 0 + assert fingerprint._link_tier({"sources": 1}, 0) == 0 + + def test_a_placeholder_artist_agrees_with_nothing(self): + track = _Track(artist="Unknown Artist", + location="/m/Unknown Artist/Unknown Album/blurf.mp3") + tokens = fingerprint._artist_tokens(artist_hint_for(track)) + assert tokens == set() + + def test_duration_still_separates_takes_by_one_artist(self): + """Artist agreement ties between two takes by the same person, so + length decides — the round 46 rule, intact.""" + payload = _payload( + _recording("Snowfall", "Ahmad Jamal", duration=155.0, sources=50, + groups=[_group("The Long Take", 2000)]), + _recording("Snowfall", "Ahmad Jamal", duration=150.3, sources=50, + groups=[_group("The Short Take", 2000)]), + ) + assert parse_lookup(payload, duration=150, + artist_hint="Ahmad Jamal")[0].album == \ + "The Short Take" + + +class TestArtistHint: + def test_it_gathers_tags_and_the_filename(self): + track = _Track(artist="dionne farris", album_artist="Dionne Farris", + location="/m/x/y/Dionne Farris - I Know [fqng9NDqKB8].mp3") + assert artist_hint_for(track).lower().count("dionne") == 3 + + def test_a_track_with_no_artist_anywhere_hints_nothing(self): + assert artist_hint_for(_Track()) == "" + + def test_the_title_is_not_part_of_it(self): + """Only the artist: every candidate here is titled "I Know", so title + words separate nothing and would just dilute the agreement count.""" + assert "know" not in artist_hint_for(DIONNE_TRACK).lower() + + +class TestTheLookupAsksForSources: + def test_meta_requests_submission_counts(self, monkeypatch): + sent = {} + + class FakeResponse: + def raise_for_status(self): pass + def json(self): return _i_know_payload() + + import requests + monkeypatch.setattr(requests, "post", + lambda *a, **k: sent.update(k["data"]) + or FakeResponse()) + candidates = lookup_fingerprint("key", 224, "fp", + hint="I Know dionne farris", + artist_hint="dionne farris") + assert "sources" in sent["meta"] + assert candidates[0].artist == "Dionne Farris" + + +@pytest.mark.parametrize("artist, expected", [ + ("Dionne Farris", 2), + ("Jay-Z", 0), + ("", 0), +]) +def test_agreement_counts_shared_artist_words(artist, expected): + candidate = fingerprint.IdentifyCandidate(score=1.0, artist="Dionne Farris") + tokens = fingerprint._artist_tokens(artist) + assert fingerprint._artist_agreement(tokens, candidate) == expected