Files
lintunes/lintunes/fingerprint.py
T
travandClaude Opus 5 c797cf570c v0.20.1: one fingerprint, six artists
trav's "Dionne Farris - I Know" kept identifying as Jay-Z, against ID3 tags
that plainly said otherwise. AcoustID was right; the ranking threw the answer
away.

The fingerprint matched one AcoustID result at 0.97, and six recordings hang
off it: Dionne Farris twice, plus Jay-Z, Marisela, New Atlantic and David
Essex, all of whom recorded a song called "I Know". A result's score belongs
to the *audio*, so every linked recording carries it however wrong the link
is. With the scores tied, ranking fell through to the duration bucket, where
Jay-Z's 222.7 s beat Dionne's 227.3 s against a 224 s file. The tags never got
a vote: the hint sat below duration in the sort key.

So the lookup now asks who submitted each link. `sources` joins LOOKUP_META —
475 people linked that audio to Dionne Farris, 6 to Jay-Z, 1 each to the rest
— and _link_tier sinks anything under a tenth of the strongest link in the
same result. The share is relative, never an absolute count, and a missing
count ranks as real: an obscure song's true link may have two submissions
against a stray's one, and rounds 45-46's payloads rank unchanged.

And it asks what the file already says. artist_hint_for gathers the artist
tag, the album artist and the artist in the filename; _artist_agreement counts
the words shared with a candidate's credit, placeholders dropped. Like every
hint since round 46 it only chooses among what AcoustID returned.

New key order: stray tier, artist agreement, duration bucket, hint overlap,
release rank — who, which take, which release. Artist above duration is the
whole fix; duration still separates two takes by one artist.

Verified live against the reported file: the proposal is now I Know — Dionne
Farris — Wild Seed - Wild Flower (1994), track 1, with all five mis-tagged
artists off the dropdown. The real response is pinned in tests/test_round52.py.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SKXUgsBBwe3qaHEjeV8ubP
2026-09-11 20:41:33 -05:00

478 lines
20 KiB
Python

"""Track identification via Chromaprint (fpcalc) + the AcoustID web API.
Pure parsing/ranking helpers are separated from the subprocess and network
calls so they can be tested offline with canned AcoustID JSON;
``TrackIdentifier`` runs the whole fingerprint+lookup on a daemon thread (the
art_search pattern) and reports back over a Qt signal, which is delivered
queued on the GUI thread.
fpcalc is the Chromaprint *CLI binary* (Fedora: chromaprint-tools, Debian:
libchromaprint-tools) — detected at runtime, never a pip dependency. The
AcoustID application key is user-supplied via Preferences (the Last.fm
precedent); registration is free at https://acoustid.org/new-application.
"""
import json
import re
import shutil
import subprocess
import threading
from dataclasses import asdict, dataclass
from pathlib import Path
from PyQt6.QtCore import QObject, pyqtSignal
from lintunes import filename_tags
ACOUSTID_LOOKUP_URL = "https://api.acoustid.org/v2/lookup"
LOOKUP_META = "recordings releasegroups releases tracks sources compress"
SCORE_THRESHOLD = 0.5 # AcoustID scores below this are noise
MAX_CANDIDATES = 8
TIMEOUT_S = 15
FPCALC_TIMEOUT_S = 60
@dataclass
class IdentifyCandidate:
"""One proposed identification. Field names match Track/tagging so
``fields()`` can go straight into ``LibraryManager.edit_track_fields``."""
score: float
# "acoustid" (a fingerprint match, so ``score`` is a real confidence) or
# "filename" (a guess read off the file's own name — no confidence to
# quote, and the dialog says so rather than implying one).
source: str = "acoustid"
name: str | None = None
artist: str | None = None
album_artist: str | None = None
album: str | None = None
year: int | None = None
track_number: int | None = None
track_count: int | None = None
disc_number: int | None = None
def fields(self) -> dict:
"""Proposed edits keyed by LinTunes field names, Nones omitted."""
return {k: v for k, v in asdict(self).items()
if k not in ("score", "source") and v is not None}
# ---- pure parsing/ranking helpers (offline-testable) ----
def _join_artists(artists: list | None) -> str | None:
"""Concatenate MusicBrainz artist credits, honoring ``joinphrase``."""
if not artists:
return None
parts = []
for i, artist in enumerate(artists):
name = artist.get("name", "")
if not name:
continue
if parts:
parts.append(artists[i - 1].get("joinphrase") or "; ")
parts.append(name)
return "".join(parts) or None
def _release_year(release: dict) -> int | None:
year = (release.get("date") or {}).get("year")
return year if isinstance(year, int) and year > 0 else None
def _oldest_year(releasegroup: dict) -> int | None:
"""Earliest dated release in the group — the original pressing, not the
CD reissue."""
years = [y for r in releasegroup.get("releases", [])
if (y := _release_year(r)) is not None]
return min(years) if years else None
# Secondary types that mean "this is not the album the song came from".
# Live is deliberately absent: a live album is a real album, and demoting it
# buried "Ahmad Jamal's Alhambra" under a compilation for a file whose own
# name said "Live At The Alhambra".
_DEMOTED_SECONDARY = {"compilation", "interview", "remix", "dj-mix",
"mixtape/street", "demo", "audiobook"}
def _releasegroup_sort_key(releasegroup: dict) -> tuple:
"""Rank a release group for the default proposal: a proper album first,
then EPs/Singles, with compilations and the like pushed down."""
rg_type = releasegroup.get("type") or ""
secondary = releasegroup.get("secondarytypes") or []
rank = {"Album": 0, "EP": 1, "Single": 2}.get(rg_type, 4)
if any((s or "").casefold() in _DEMOTED_SECONDARY for s in secondary):
rank += 3
return (rank, _oldest_year(releasegroup) or 9999)
_TOKEN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE)
_STOPWORDS = {"the", "a", "an", "of", "and", "in", "at", "on", "live",
"feat", "ft", "remastered", "version", "mix", "edit"}
def _tokens(text: str) -> set:
"""Comparable words from a title/album/filename."""
return {t.casefold() for t in _TOKEN.findall(text or "")
if len(t) > 1 and t.casefold() not in _STOPWORDS}
def _duration_bucket(recording: dict, duration: float | None) -> int:
"""How well a recording's length matches the file's, coarsely.
AcoustID often returns several recordings of the same song — a studio
take, a live take, an edit — and their lengths are what tell them apart.
Buckets rather than raw difference so a second or two never outweighs the
other evidence, but a clearly different take sinks.
"""
other = recording.get("duration")
if duration is None or not isinstance(other, (int, float)):
return 1 # unknown: behind an exact match, ahead of a bad one
diff = abs(other - duration)
if diff <= 1.5:
return 0
if diff <= 4:
return 1
if diff <= 12:
return 2
return 3
def _hint_overlap(hint_tokens: set, candidate: IdentifyCandidate) -> int:
"""How much a candidate echoes what the file already claims to be.
The distinctive words in "Snowfall (Live At The Alhambra_1961)" —
*snowfall*, *alhambra*, *1961* — are the best evidence available about
which of eight plausible releases this actually is, and ignoring them was
why a compilation could outrank the album named in the filename.
"""
if not hint_tokens:
return 0
text = " ".join(p for p in (candidate.name, candidate.artist,
candidate.album_artist, candidate.album) if p)
# A word like "alhambra" names a release; a bare number like "1961" turns
# up in every compilation spanning that year, so it counts for less.
return sum(2 if not token.isdigit() else 1
for token in hint_tokens & _tokens(text))
# A link backed by fewer than this share of the strongest link's submissions
# is a stray rather than a rival: 6 against 475 is somebody's mis-tag, while
# 3 against 20 is a genuinely contested song.
STRAY_LINK_SHARE = 0.1
def _link_tier(recording: dict, strongest: int) -> int:
"""0 for a real link, 1 for a stray, by how many people submitted it.
One AcoustID result is one piece of *audio*, and its score says how well
the file matched that audio — so every recording linked to it shares that
score, however wrong the link is. Links are user-submitted, and a song
whose title another artist also used collects mis-tags: trav's Dionne
Farris "I Know" is linked to Jay-Z's "I Know", Marisela's, David Essex's
and New Atlantic's. What separates them is how many people submitted each
link — 475 against 6, 1, 1 and 1 — which is the only field in the response
that knows the difference.
Relative to the strongest link, never an absolute count: an obscure song's
real link may have two submissions against a stray's one. Unknown counts
(an older payload, or a result where nobody reported any) rank as real, so
nothing is demoted on missing evidence.
"""
sources = recording.get("sources")
if not isinstance(sources, int) or strongest <= 0:
return 0
return 1 if sources < strongest * STRAY_LINK_SHARE else 0
def _strongest_link(result: dict) -> int:
"""The most submissions behind any one link of an AcoustID result."""
counts = [r.get("sources") for r in result.get("recordings", [])]
return max((c for c in counts if isinstance(c, int)), default=0)
# Words that name no artist, so agreeing with them means nothing.
_PLACEHOLDER_ARTIST = {"unknown", "various", "artist", "artists"}
def _artist_tokens(text: str) -> set:
"""Comparable words from an artist credit, placeholders dropped."""
return {t for t in _tokens(text) if not t.isdigit()} - _PLACEHOLDER_ARTIST
def _artist_agreement(artist_tokens: set, candidate: IdentifyCandidate) -> int:
"""How much a candidate's artist echoes the one the file already names.
Only ever chooses *among* the recordings AcoustID linked to this audio, so
like the rest of the hint it can never invent a candidate — it just stops a
file whose tags plainly say "dionne farris" from being told it is Jay-Z.
"""
if not artist_tokens:
return 0
text = " ".join(p for p in (candidate.artist, candidate.album_artist) if p)
return len(artist_tokens & _artist_tokens(text))
def _pick_release(releasegroup: dict) -> dict | None:
"""Earliest dated release that carries mediums (track positions); falls
back to the earliest dated one, then the first."""
releases = releasegroup.get("releases", [])
if not releases:
return None
dated = sorted(releases, key=lambda r: _release_year(r) or 9999)
for release in dated:
if release.get("mediums"):
return release
return dated[0]
def _track_position(release: dict | None) -> tuple[int | None, int | None,
int | None]:
"""(track_number, track_count, disc_number) from the matching medium.
With ``meta=tracks`` AcoustID returns only the medium/track entries that
match the looked-up recording, so the first entries are the match.
"""
if not release:
return None, None, None
mediums = release.get("mediums") or []
if not mediums:
return None, None, None
medium = mediums[0]
tracks = medium.get("tracks") or []
number = tracks[0].get("position") if tracks else None
count = medium.get("track_count")
disc = medium.get("position")
return (number if isinstance(number, int) else None,
count if isinstance(count, int) else None,
disc if isinstance(disc, int) else None)
def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD,
limit: int = MAX_CANDIDATES, hint: str = "",
duration: float | None = None,
artist_hint: str = "") -> list[IdentifyCandidate]:
"""Turn an AcoustID lookup response into ranked candidates.
One candidate per release group of each matched recording, so the
alternates offered in the dialog genuinely differ (different album/year).
Every candidate of a recording carries the recording's *original* year —
the minimum across all its release groups' releases — even when the
proposed album is a later compilation: years sort by when the song first
came out, not by which pressing this file happens to be from.
``hint`` is whatever the file already claims to be (its tags and its
filename), and ``artist_hint`` is just the artist part of that. Neither
invents a candidate; they decide between the ones AcoustID returned.
The order is *who*, then *which take*, then *which release*: a stray link
(see ``_link_tier``) sinks below the links people actually submitted, then
an artist agreeing with the file's own tags wins, then the recording whose
length matches, and only then the release whose words echo the hint. Artist
before duration is the Dionne Farris rule — five artists' mis-tags hang off
that one fingerprint, and Jay-Z's "I Know" happened to be 1.5 s closer.
A year named in the hint still wins when it predates anything the database
knows: MusicBrainz often holds only a reissue's date, with the original
pressing undated.
"""
hint_tokens = _tokens(hint)
hint_year = _single_year(hint)
artist_tokens = _artist_tokens(artist_hint)
scored = []
seen = set()
for result in payload.get("results", []):
score = result.get("score", 0)
if score < threshold:
continue
strongest = _strongest_link(result)
for recording in result.get("recordings", []):
tier = _link_tier(recording, strongest)
length = _duration_bucket(recording, duration)
title = recording.get("title")
artist = _join_artists(recording.get("artists"))
groups = sorted(recording.get("releasegroups") or [],
key=_releasegroup_sort_key)
original_years = [y for rg in groups
if (y := _oldest_year(rg)) is not None]
original_year = min(original_years) if original_years else None
if hint_year is not None and (original_year is None
or hint_year < original_year):
original_year = hint_year
if not groups:
if title:
bare = IdentifyCandidate(score=score, name=title,
artist=artist, year=hint_year)
scored.append(((-round(score, 2), tier,
-_artist_agreement(artist_tokens, bare),
length,
-_hint_overlap(hint_tokens, bare),
4, 9999), bare))
continue
for releasegroup in groups:
album = releasegroup.get("title")
key = (title, artist, album)
if key in seen:
continue
seen.add(key)
number, count, disc = _track_position(
_pick_release(releasegroup))
candidate = IdentifyCandidate(
score=score,
name=title,
artist=artist,
album_artist=_join_artists(releasegroup.get("artists")),
album=album,
year=original_year,
track_number=number,
track_count=count,
disc_number=disc,
)
rank, rg_year = _releasegroup_sort_key(releasegroup)
scored.append(((-round(score, 2), tier,
-_artist_agreement(artist_tokens, candidate),
length,
-_hint_overlap(hint_tokens, candidate),
rank, rg_year), candidate))
scored.sort(key=lambda pair: pair[0])
return [candidate for _key, candidate in scored][:limit]
def _single_year(text: str) -> int | None:
"""The year a string names, when it names exactly one (so a range like
"1958-62" is never mistaken for a release year)."""
return filename_tags.year_in(text or "")
def hint_for(track) -> str:
"""What a track already claims to be — its tags plus its filename. Used
to rank lookup results, and to fall back on when nothing matches."""
parts = [track.name or "", track.artist or "", track.album or ""]
if track.location:
parts.append(Path(track.location).stem)
return " ".join(p for p in parts if p)
def artist_hint_for(track) -> str:
"""Who the file already says made this — its artist tags plus the artist
its own name carries. Ranks the recordings linked to one fingerprint."""
parts = [track.artist or "", getattr(track, "album_artist", "") or ""]
if track.location:
parts.append(filename_tags.parse_filename(track.location)
.get("artist", ""))
return " ".join(p for p in parts if p)
def candidate_from_filename(track) -> IdentifyCandidate | None:
"""A proposal read off the file's own name, for the very common case that
no database has ever heard of the song.
Underground and self-released music simply isn't in AcoustID — the
fingerprint is fine, there is just nothing to match it against — but the
download named the file "Artist - Title [id]", which is how a human reads
it too. Returns None when the name yields nothing usable.
"""
if not track.location:
return None
fields = filename_tags.parse_filename(track.location)
candidate = IdentifyCandidate(score=0.0, source="filename")
for key, value in fields.items():
if hasattr(candidate, key):
setattr(candidate, key, value)
return candidate if candidate.fields() else None
# ---- subprocess + network ----
def fpcalc_available() -> bool:
"""Whether the fpcalc *binary* is on PATH (the ffmpeg_available pattern —
Chromaprint's CLI tool, not a Python package)."""
return shutil.which("fpcalc") is not None
def fingerprint_file(location: str) -> tuple[int, str]:
"""Run fpcalc on a file; returns (duration seconds, fingerprint).
Raises OSError on failure, carrying fpcalc's last stderr line."""
try:
proc = subprocess.run(
["fpcalc", "-json", location],
stdin=subprocess.DEVNULL, capture_output=True,
timeout=FPCALC_TIMEOUT_S)
except (OSError, ValueError, subprocess.TimeoutExpired) as e:
raise OSError(f"couldn't run fpcalc: {e}") from e
name = location.rsplit("/", 1)[-1]
if proc.returncode != 0:
detail = proc.stderr.decode("utf-8", "replace").strip().splitlines()
raise OSError(f"fpcalc failed on {name}"
+ (f": {detail[-1]}" if detail else ""))
try:
payload = json.loads(proc.stdout.decode("utf-8", "replace"))
return int(payload["duration"]), payload["fingerprint"]
except (ValueError, KeyError, TypeError) as e:
raise OSError(f"couldn't read fpcalc output for {name}") from e
def lookup_fingerprint(api_key: str, duration: int, fingerprint: str,
hint: str = "",
artist_hint: str = "") -> list[IdentifyCandidate]:
import requests
response = requests.post(
ACOUSTID_LOOKUP_URL,
data={"client": api_key, "duration": duration,
"fingerprint": fingerprint, "meta": LOOKUP_META,
"format": "json"},
timeout=TIMEOUT_S,
)
response.raise_for_status()
payload = response.json()
if payload.get("status") != "ok":
message = (payload.get("error") or {}).get("message", "lookup failed")
raise RuntimeError(message)
return parse_lookup(payload, hint=hint, duration=duration,
artist_hint=artist_hint)
class TrackIdentifier(QObject):
"""One fingerprint+lookup running off the GUI thread.
``finished`` carries {"track_id": int, "candidates": [IdentifyCandidate]}
or {"track_id": int, "error": str}. A filename guess is appended to the
fingerprint's matches, and stands alone when there are none — a lookup
that found nothing is not the same as nothing to propose.
"""
finished = pyqtSignal(object)
def identify(self, track, api_key: str):
track_id = track.track_id
location = track.location
hint = hint_for(track)
artist_hint = artist_hint_for(track)
guess = candidate_from_filename(track)
def work():
try:
# No key: there's nobody to ask, so skip fpcalc and the
# network entirely and fall through to the filename. Only an
# Import from URL reaches here, because the menu route insists
# on setup first. A missing fpcalc lands in the same place,
# via fingerprint_file raising.
if not api_key:
raise RuntimeError("no AcoustID key set")
duration, fp = fingerprint_file(location)
candidates = lookup_fingerprint(api_key, duration, fp, hint,
artist_hint)
except Exception as e:
# A file too short or too odd to fingerprint still has a name.
if guess is not None:
self.finished.emit({"track_id": track_id,
"candidates": [guess],
"note": str(e)})
else:
self.finished.emit({"track_id": track_id, "error": str(e)})
return
if guess is not None:
candidates = candidates + [guess]
self.finished.emit(
{"track_id": track_id, "candidates": candidates})
threading.Thread(target=work, daemon=True).start()