trav's "Dionne Farris - I Know" kept identifying as Jay-Z, against ID3 tags that plainly said otherwise. AcoustID was right; the ranking threw the answer away. The fingerprint matched one AcoustID result at 0.97, and six recordings hang off it: Dionne Farris twice, plus Jay-Z, Marisela, New Atlantic and David Essex, all of whom recorded a song called "I Know". A result's score belongs to the *audio*, so every linked recording carries it however wrong the link is. With the scores tied, ranking fell through to the duration bucket, where Jay-Z's 222.7 s beat Dionne's 227.3 s against a 224 s file. The tags never got a vote: the hint sat below duration in the sort key. So the lookup now asks who submitted each link. `sources` joins LOOKUP_META — 475 people linked that audio to Dionne Farris, 6 to Jay-Z, 1 each to the rest — and _link_tier sinks anything under a tenth of the strongest link in the same result. The share is relative, never an absolute count, and a missing count ranks as real: an obscure song's true link may have two submissions against a stray's one, and rounds 45-46's payloads rank unchanged. And it asks what the file already says. artist_hint_for gathers the artist tag, the album artist and the artist in the filename; _artist_agreement counts the words shared with a candidate's credit, placeholders dropped. Like every hint since round 46 it only chooses among what AcoustID returned. New key order: stray tier, artist agreement, duration bucket, hint overlap, release rank — who, which take, which release. Artist above duration is the whole fix; duration still separates two takes by one artist. Verified live against the reported file: the proposal is now I Know — Dionne Farris — Wild Seed - Wild Flower (1994), track 1, with all five mis-tagged artists off the dropdown. The real response is pinned in tests/test_round52.py. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SKXUgsBBwe3qaHEjeV8ubP
478 lines
20 KiB
Python
478 lines
20 KiB
Python
"""Track identification via Chromaprint (fpcalc) + the AcoustID web API.
|
|
|
|
Pure parsing/ranking helpers are separated from the subprocess and network
|
|
calls so they can be tested offline with canned AcoustID JSON;
|
|
``TrackIdentifier`` runs the whole fingerprint+lookup on a daemon thread (the
|
|
art_search pattern) and reports back over a Qt signal, which is delivered
|
|
queued on the GUI thread.
|
|
|
|
fpcalc is the Chromaprint *CLI binary* (Fedora: chromaprint-tools, Debian:
|
|
libchromaprint-tools) — detected at runtime, never a pip dependency. The
|
|
AcoustID application key is user-supplied via Preferences (the Last.fm
|
|
precedent); registration is free at https://acoustid.org/new-application.
|
|
"""
|
|
import json
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import threading
|
|
from dataclasses import asdict, dataclass
|
|
from pathlib import Path
|
|
|
|
from PyQt6.QtCore import QObject, pyqtSignal
|
|
|
|
from lintunes import filename_tags
|
|
|
|
|
|
ACOUSTID_LOOKUP_URL = "https://api.acoustid.org/v2/lookup"
|
|
LOOKUP_META = "recordings releasegroups releases tracks sources compress"
|
|
SCORE_THRESHOLD = 0.5 # AcoustID scores below this are noise
|
|
MAX_CANDIDATES = 8
|
|
TIMEOUT_S = 15
|
|
FPCALC_TIMEOUT_S = 60
|
|
|
|
|
|
@dataclass
|
|
class IdentifyCandidate:
|
|
"""One proposed identification. Field names match Track/tagging so
|
|
``fields()`` can go straight into ``LibraryManager.edit_track_fields``."""
|
|
score: float
|
|
# "acoustid" (a fingerprint match, so ``score`` is a real confidence) or
|
|
# "filename" (a guess read off the file's own name — no confidence to
|
|
# quote, and the dialog says so rather than implying one).
|
|
source: str = "acoustid"
|
|
name: str | None = None
|
|
artist: str | None = None
|
|
album_artist: str | None = None
|
|
album: str | None = None
|
|
year: int | None = None
|
|
track_number: int | None = None
|
|
track_count: int | None = None
|
|
disc_number: int | None = None
|
|
|
|
def fields(self) -> dict:
|
|
"""Proposed edits keyed by LinTunes field names, Nones omitted."""
|
|
return {k: v for k, v in asdict(self).items()
|
|
if k not in ("score", "source") and v is not None}
|
|
|
|
|
|
# ---- pure parsing/ranking helpers (offline-testable) ----
|
|
|
|
def _join_artists(artists: list | None) -> str | None:
|
|
"""Concatenate MusicBrainz artist credits, honoring ``joinphrase``."""
|
|
if not artists:
|
|
return None
|
|
parts = []
|
|
for i, artist in enumerate(artists):
|
|
name = artist.get("name", "")
|
|
if not name:
|
|
continue
|
|
if parts:
|
|
parts.append(artists[i - 1].get("joinphrase") or "; ")
|
|
parts.append(name)
|
|
return "".join(parts) or None
|
|
|
|
|
|
def _release_year(release: dict) -> int | None:
|
|
year = (release.get("date") or {}).get("year")
|
|
return year if isinstance(year, int) and year > 0 else None
|
|
|
|
|
|
def _oldest_year(releasegroup: dict) -> int | None:
|
|
"""Earliest dated release in the group — the original pressing, not the
|
|
CD reissue."""
|
|
years = [y for r in releasegroup.get("releases", [])
|
|
if (y := _release_year(r)) is not None]
|
|
return min(years) if years else None
|
|
|
|
|
|
# Secondary types that mean "this is not the album the song came from".
|
|
# Live is deliberately absent: a live album is a real album, and demoting it
|
|
# buried "Ahmad Jamal's Alhambra" under a compilation for a file whose own
|
|
# name said "Live At The Alhambra".
|
|
_DEMOTED_SECONDARY = {"compilation", "interview", "remix", "dj-mix",
|
|
"mixtape/street", "demo", "audiobook"}
|
|
|
|
|
|
def _releasegroup_sort_key(releasegroup: dict) -> tuple:
|
|
"""Rank a release group for the default proposal: a proper album first,
|
|
then EPs/Singles, with compilations and the like pushed down."""
|
|
rg_type = releasegroup.get("type") or ""
|
|
secondary = releasegroup.get("secondarytypes") or []
|
|
rank = {"Album": 0, "EP": 1, "Single": 2}.get(rg_type, 4)
|
|
if any((s or "").casefold() in _DEMOTED_SECONDARY for s in secondary):
|
|
rank += 3
|
|
return (rank, _oldest_year(releasegroup) or 9999)
|
|
|
|
|
|
_TOKEN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE)
|
|
_STOPWORDS = {"the", "a", "an", "of", "and", "in", "at", "on", "live",
|
|
"feat", "ft", "remastered", "version", "mix", "edit"}
|
|
|
|
|
|
def _tokens(text: str) -> set:
|
|
"""Comparable words from a title/album/filename."""
|
|
return {t.casefold() for t in _TOKEN.findall(text or "")
|
|
if len(t) > 1 and t.casefold() not in _STOPWORDS}
|
|
|
|
|
|
def _duration_bucket(recording: dict, duration: float | None) -> int:
|
|
"""How well a recording's length matches the file's, coarsely.
|
|
|
|
AcoustID often returns several recordings of the same song — a studio
|
|
take, a live take, an edit — and their lengths are what tell them apart.
|
|
Buckets rather than raw difference so a second or two never outweighs the
|
|
other evidence, but a clearly different take sinks.
|
|
"""
|
|
other = recording.get("duration")
|
|
if duration is None or not isinstance(other, (int, float)):
|
|
return 1 # unknown: behind an exact match, ahead of a bad one
|
|
diff = abs(other - duration)
|
|
if diff <= 1.5:
|
|
return 0
|
|
if diff <= 4:
|
|
return 1
|
|
if diff <= 12:
|
|
return 2
|
|
return 3
|
|
|
|
|
|
def _hint_overlap(hint_tokens: set, candidate: IdentifyCandidate) -> int:
|
|
"""How much a candidate echoes what the file already claims to be.
|
|
|
|
The distinctive words in "Snowfall (Live At The Alhambra_1961)" —
|
|
*snowfall*, *alhambra*, *1961* — are the best evidence available about
|
|
which of eight plausible releases this actually is, and ignoring them was
|
|
why a compilation could outrank the album named in the filename.
|
|
"""
|
|
if not hint_tokens:
|
|
return 0
|
|
text = " ".join(p for p in (candidate.name, candidate.artist,
|
|
candidate.album_artist, candidate.album) if p)
|
|
# A word like "alhambra" names a release; a bare number like "1961" turns
|
|
# up in every compilation spanning that year, so it counts for less.
|
|
return sum(2 if not token.isdigit() else 1
|
|
for token in hint_tokens & _tokens(text))
|
|
|
|
|
|
# A link backed by fewer than this share of the strongest link's submissions
|
|
# is a stray rather than a rival: 6 against 475 is somebody's mis-tag, while
|
|
# 3 against 20 is a genuinely contested song.
|
|
STRAY_LINK_SHARE = 0.1
|
|
|
|
|
|
def _link_tier(recording: dict, strongest: int) -> int:
|
|
"""0 for a real link, 1 for a stray, by how many people submitted it.
|
|
|
|
One AcoustID result is one piece of *audio*, and its score says how well
|
|
the file matched that audio — so every recording linked to it shares that
|
|
score, however wrong the link is. Links are user-submitted, and a song
|
|
whose title another artist also used collects mis-tags: trav's Dionne
|
|
Farris "I Know" is linked to Jay-Z's "I Know", Marisela's, David Essex's
|
|
and New Atlantic's. What separates them is how many people submitted each
|
|
link — 475 against 6, 1, 1 and 1 — which is the only field in the response
|
|
that knows the difference.
|
|
|
|
Relative to the strongest link, never an absolute count: an obscure song's
|
|
real link may have two submissions against a stray's one. Unknown counts
|
|
(an older payload, or a result where nobody reported any) rank as real, so
|
|
nothing is demoted on missing evidence.
|
|
"""
|
|
sources = recording.get("sources")
|
|
if not isinstance(sources, int) or strongest <= 0:
|
|
return 0
|
|
return 1 if sources < strongest * STRAY_LINK_SHARE else 0
|
|
|
|
|
|
def _strongest_link(result: dict) -> int:
|
|
"""The most submissions behind any one link of an AcoustID result."""
|
|
counts = [r.get("sources") for r in result.get("recordings", [])]
|
|
return max((c for c in counts if isinstance(c, int)), default=0)
|
|
|
|
|
|
# Words that name no artist, so agreeing with them means nothing.
|
|
_PLACEHOLDER_ARTIST = {"unknown", "various", "artist", "artists"}
|
|
|
|
|
|
def _artist_tokens(text: str) -> set:
|
|
"""Comparable words from an artist credit, placeholders dropped."""
|
|
return {t for t in _tokens(text) if not t.isdigit()} - _PLACEHOLDER_ARTIST
|
|
|
|
|
|
def _artist_agreement(artist_tokens: set, candidate: IdentifyCandidate) -> int:
|
|
"""How much a candidate's artist echoes the one the file already names.
|
|
|
|
Only ever chooses *among* the recordings AcoustID linked to this audio, so
|
|
like the rest of the hint it can never invent a candidate — it just stops a
|
|
file whose tags plainly say "dionne farris" from being told it is Jay-Z.
|
|
"""
|
|
if not artist_tokens:
|
|
return 0
|
|
text = " ".join(p for p in (candidate.artist, candidate.album_artist) if p)
|
|
return len(artist_tokens & _artist_tokens(text))
|
|
|
|
|
|
def _pick_release(releasegroup: dict) -> dict | None:
|
|
"""Earliest dated release that carries mediums (track positions); falls
|
|
back to the earliest dated one, then the first."""
|
|
releases = releasegroup.get("releases", [])
|
|
if not releases:
|
|
return None
|
|
dated = sorted(releases, key=lambda r: _release_year(r) or 9999)
|
|
for release in dated:
|
|
if release.get("mediums"):
|
|
return release
|
|
return dated[0]
|
|
|
|
|
|
def _track_position(release: dict | None) -> tuple[int | None, int | None,
|
|
int | None]:
|
|
"""(track_number, track_count, disc_number) from the matching medium.
|
|
|
|
With ``meta=tracks`` AcoustID returns only the medium/track entries that
|
|
match the looked-up recording, so the first entries are the match.
|
|
"""
|
|
if not release:
|
|
return None, None, None
|
|
mediums = release.get("mediums") or []
|
|
if not mediums:
|
|
return None, None, None
|
|
medium = mediums[0]
|
|
tracks = medium.get("tracks") or []
|
|
number = tracks[0].get("position") if tracks else None
|
|
count = medium.get("track_count")
|
|
disc = medium.get("position")
|
|
return (number if isinstance(number, int) else None,
|
|
count if isinstance(count, int) else None,
|
|
disc if isinstance(disc, int) else None)
|
|
|
|
|
|
def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD,
|
|
limit: int = MAX_CANDIDATES, hint: str = "",
|
|
duration: float | None = None,
|
|
artist_hint: str = "") -> list[IdentifyCandidate]:
|
|
"""Turn an AcoustID lookup response into ranked candidates.
|
|
|
|
One candidate per release group of each matched recording, so the
|
|
alternates offered in the dialog genuinely differ (different album/year).
|
|
Every candidate of a recording carries the recording's *original* year —
|
|
the minimum across all its release groups' releases — even when the
|
|
proposed album is a later compilation: years sort by when the song first
|
|
came out, not by which pressing this file happens to be from.
|
|
|
|
``hint`` is whatever the file already claims to be (its tags and its
|
|
filename), and ``artist_hint`` is just the artist part of that. Neither
|
|
invents a candidate; they decide between the ones AcoustID returned.
|
|
|
|
The order is *who*, then *which take*, then *which release*: a stray link
|
|
(see ``_link_tier``) sinks below the links people actually submitted, then
|
|
an artist agreeing with the file's own tags wins, then the recording whose
|
|
length matches, and only then the release whose words echo the hint. Artist
|
|
before duration is the Dionne Farris rule — five artists' mis-tags hang off
|
|
that one fingerprint, and Jay-Z's "I Know" happened to be 1.5 s closer.
|
|
|
|
A year named in the hint still wins when it predates anything the database
|
|
knows: MusicBrainz often holds only a reissue's date, with the original
|
|
pressing undated.
|
|
"""
|
|
hint_tokens = _tokens(hint)
|
|
hint_year = _single_year(hint)
|
|
artist_tokens = _artist_tokens(artist_hint)
|
|
scored = []
|
|
seen = set()
|
|
for result in payload.get("results", []):
|
|
score = result.get("score", 0)
|
|
if score < threshold:
|
|
continue
|
|
strongest = _strongest_link(result)
|
|
for recording in result.get("recordings", []):
|
|
tier = _link_tier(recording, strongest)
|
|
length = _duration_bucket(recording, duration)
|
|
title = recording.get("title")
|
|
artist = _join_artists(recording.get("artists"))
|
|
groups = sorted(recording.get("releasegroups") or [],
|
|
key=_releasegroup_sort_key)
|
|
original_years = [y for rg in groups
|
|
if (y := _oldest_year(rg)) is not None]
|
|
original_year = min(original_years) if original_years else None
|
|
if hint_year is not None and (original_year is None
|
|
or hint_year < original_year):
|
|
original_year = hint_year
|
|
if not groups:
|
|
if title:
|
|
bare = IdentifyCandidate(score=score, name=title,
|
|
artist=artist, year=hint_year)
|
|
scored.append(((-round(score, 2), tier,
|
|
-_artist_agreement(artist_tokens, bare),
|
|
length,
|
|
-_hint_overlap(hint_tokens, bare),
|
|
4, 9999), bare))
|
|
continue
|
|
for releasegroup in groups:
|
|
album = releasegroup.get("title")
|
|
key = (title, artist, album)
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
number, count, disc = _track_position(
|
|
_pick_release(releasegroup))
|
|
candidate = IdentifyCandidate(
|
|
score=score,
|
|
name=title,
|
|
artist=artist,
|
|
album_artist=_join_artists(releasegroup.get("artists")),
|
|
album=album,
|
|
year=original_year,
|
|
track_number=number,
|
|
track_count=count,
|
|
disc_number=disc,
|
|
)
|
|
rank, rg_year = _releasegroup_sort_key(releasegroup)
|
|
scored.append(((-round(score, 2), tier,
|
|
-_artist_agreement(artist_tokens, candidate),
|
|
length,
|
|
-_hint_overlap(hint_tokens, candidate),
|
|
rank, rg_year), candidate))
|
|
scored.sort(key=lambda pair: pair[0])
|
|
return [candidate for _key, candidate in scored][:limit]
|
|
|
|
|
|
def _single_year(text: str) -> int | None:
|
|
"""The year a string names, when it names exactly one (so a range like
|
|
"1958-62" is never mistaken for a release year)."""
|
|
return filename_tags.year_in(text or "")
|
|
|
|
|
|
def hint_for(track) -> str:
|
|
"""What a track already claims to be — its tags plus its filename. Used
|
|
to rank lookup results, and to fall back on when nothing matches."""
|
|
parts = [track.name or "", track.artist or "", track.album or ""]
|
|
if track.location:
|
|
parts.append(Path(track.location).stem)
|
|
return " ".join(p for p in parts if p)
|
|
|
|
|
|
def artist_hint_for(track) -> str:
|
|
"""Who the file already says made this — its artist tags plus the artist
|
|
its own name carries. Ranks the recordings linked to one fingerprint."""
|
|
parts = [track.artist or "", getattr(track, "album_artist", "") or ""]
|
|
if track.location:
|
|
parts.append(filename_tags.parse_filename(track.location)
|
|
.get("artist", ""))
|
|
return " ".join(p for p in parts if p)
|
|
|
|
|
|
def candidate_from_filename(track) -> IdentifyCandidate | None:
|
|
"""A proposal read off the file's own name, for the very common case that
|
|
no database has ever heard of the song.
|
|
|
|
Underground and self-released music simply isn't in AcoustID — the
|
|
fingerprint is fine, there is just nothing to match it against — but the
|
|
download named the file "Artist - Title [id]", which is how a human reads
|
|
it too. Returns None when the name yields nothing usable.
|
|
"""
|
|
if not track.location:
|
|
return None
|
|
fields = filename_tags.parse_filename(track.location)
|
|
candidate = IdentifyCandidate(score=0.0, source="filename")
|
|
for key, value in fields.items():
|
|
if hasattr(candidate, key):
|
|
setattr(candidate, key, value)
|
|
return candidate if candidate.fields() else None
|
|
|
|
|
|
# ---- subprocess + network ----
|
|
|
|
def fpcalc_available() -> bool:
|
|
"""Whether the fpcalc *binary* is on PATH (the ffmpeg_available pattern —
|
|
Chromaprint's CLI tool, not a Python package)."""
|
|
return shutil.which("fpcalc") is not None
|
|
|
|
|
|
def fingerprint_file(location: str) -> tuple[int, str]:
|
|
"""Run fpcalc on a file; returns (duration seconds, fingerprint).
|
|
Raises OSError on failure, carrying fpcalc's last stderr line."""
|
|
try:
|
|
proc = subprocess.run(
|
|
["fpcalc", "-json", location],
|
|
stdin=subprocess.DEVNULL, capture_output=True,
|
|
timeout=FPCALC_TIMEOUT_S)
|
|
except (OSError, ValueError, subprocess.TimeoutExpired) as e:
|
|
raise OSError(f"couldn't run fpcalc: {e}") from e
|
|
name = location.rsplit("/", 1)[-1]
|
|
if proc.returncode != 0:
|
|
detail = proc.stderr.decode("utf-8", "replace").strip().splitlines()
|
|
raise OSError(f"fpcalc failed on {name}"
|
|
+ (f": {detail[-1]}" if detail else ""))
|
|
try:
|
|
payload = json.loads(proc.stdout.decode("utf-8", "replace"))
|
|
return int(payload["duration"]), payload["fingerprint"]
|
|
except (ValueError, KeyError, TypeError) as e:
|
|
raise OSError(f"couldn't read fpcalc output for {name}") from e
|
|
|
|
|
|
def lookup_fingerprint(api_key: str, duration: int, fingerprint: str,
|
|
hint: str = "",
|
|
artist_hint: str = "") -> list[IdentifyCandidate]:
|
|
import requests
|
|
response = requests.post(
|
|
ACOUSTID_LOOKUP_URL,
|
|
data={"client": api_key, "duration": duration,
|
|
"fingerprint": fingerprint, "meta": LOOKUP_META,
|
|
"format": "json"},
|
|
timeout=TIMEOUT_S,
|
|
)
|
|
response.raise_for_status()
|
|
payload = response.json()
|
|
if payload.get("status") != "ok":
|
|
message = (payload.get("error") or {}).get("message", "lookup failed")
|
|
raise RuntimeError(message)
|
|
return parse_lookup(payload, hint=hint, duration=duration,
|
|
artist_hint=artist_hint)
|
|
|
|
|
|
class TrackIdentifier(QObject):
|
|
"""One fingerprint+lookup running off the GUI thread.
|
|
|
|
``finished`` carries {"track_id": int, "candidates": [IdentifyCandidate]}
|
|
or {"track_id": int, "error": str}. A filename guess is appended to the
|
|
fingerprint's matches, and stands alone when there are none — a lookup
|
|
that found nothing is not the same as nothing to propose.
|
|
"""
|
|
|
|
finished = pyqtSignal(object)
|
|
|
|
def identify(self, track, api_key: str):
|
|
track_id = track.track_id
|
|
location = track.location
|
|
hint = hint_for(track)
|
|
artist_hint = artist_hint_for(track)
|
|
guess = candidate_from_filename(track)
|
|
|
|
def work():
|
|
try:
|
|
# No key: there's nobody to ask, so skip fpcalc and the
|
|
# network entirely and fall through to the filename. Only an
|
|
# Import from URL reaches here, because the menu route insists
|
|
# on setup first. A missing fpcalc lands in the same place,
|
|
# via fingerprint_file raising.
|
|
if not api_key:
|
|
raise RuntimeError("no AcoustID key set")
|
|
duration, fp = fingerprint_file(location)
|
|
candidates = lookup_fingerprint(api_key, duration, fp, hint,
|
|
artist_hint)
|
|
except Exception as e:
|
|
# A file too short or too odd to fingerprint still has a name.
|
|
if guess is not None:
|
|
self.finished.emit({"track_id": track_id,
|
|
"candidates": [guess],
|
|
"note": str(e)})
|
|
else:
|
|
self.finished.emit({"track_id": track_id, "error": str(e)})
|
|
return
|
|
if guess is not None:
|
|
candidates = candidates + [guess]
|
|
self.finished.emit(
|
|
{"track_id": track_id, "candidates": candidates})
|
|
threading.Thread(target=work, daemon=True).start()
|