From 4acc57b64857ab69fa3f3c9f4825f0d40182e54e Mon Sep 17 00:00:00 2001 From: trav Date: Wed, 2 Sep 2026 17:24:07 -0400 Subject: [PATCH] v0.16.0: the file already told you MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two real failures from the first hour of round 45, both fixed by treating the file itself as evidence. "B. Clem - Zuuso [1025657891]" fingerprinted fine and matched nothing. Checked before building: AcoustID returns zero results, a MusicBrainz text search for "Zuuso" returns zero, iTunes returns zero. The track is in no metadata service on earth, so no smarter lookup could have answered it — but its filename said exactly what it was. New filename_tags.py reads that, validated against all 474 files in the untagged folder rather than invented examples: it strips yt-dlp ids, drops "(Official Video)" noise, splits Artist - Title, reads a leading 1-04, and falls back to the iTunes tree. It is careful about what it removes — an 11-char trailing token counts as a YouTube id only if it carries a digit, an underscore or flipping case, or "Cold Draft-Underground" loses a word, and a hyphen only splits when it has spaces around it so Jay-Z survives. A guess says it is one: candidates carry a source, and the dialog quotes a confidence only for real fingerprint matches. The file now ranks lookup results too, which fixes the Alhambra case: token overlap (words double, bare numbers single), a duration bucket against the recording lengths AcoustID returns, and a filename year that predates what the database knows. Live also stops counting as a demerit — it was lumped in with Compilation and buried a live album for a file whose name said "Live At The Alhambra". That track now proposes the right album with year 1961 instead of a compilation with 2002. The year genuinely needed the filename: MusicBrainz's own first-release-date for "Ahmad Jamal's Alhambra" is 2002, because its three original 1961 pressings are stored undated. The obvious fix of a second MusicBrainz call returns the same wrong answer. Not done on purpose: trav rejected "never propose a shorter title" — truncating is wanted, since the garbage is usually the part being dropped. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01NsiFHdyVg1UhBSxDJfTNRm --- CLAUDE.md | 28 ++++ TASKS.md | 26 ++++ lintunes/__init__.py | 2 +- lintunes/filename_tags.py | 164 +++++++++++++++++++++++ lintunes/fingerprint.py | 186 ++++++++++++++++++++++---- lintunes/gui/identify_dialog.py | 26 +++- lintunes/gui/main_window.py | 5 +- tasks-done.md | 46 +++++++ tests/test_round46.py | 227 ++++++++++++++++++++++++++++++++ 9 files changed, 674 insertions(+), 36 deletions(-) create mode 100644 lintunes/filename_tags.py create mode 100644 tests/test_round46.py diff --git a/CLAUDE.md b/CLAUDE.md index 584837b..4982bc0 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -217,6 +217,34 @@ persistence) → GUI (Qt widgets that read the manager and connect to its signal it never belongs in git). Audio under ~3 s has no fingerprint at all ("Empty fingerprint"), which is a reported failure, not a crash. +- **`lintunes/filename_tags.py`** — the offline half of Identify Track, and the + answer to its biggest limitation: **AcoustID only knows music somebody + submitted**, so an underground/SoundCloud rip fingerprints perfectly and + matches *nothing* (verified: "B. Clem — Zuuso" returns zero results from + AcoustID, zero from a MusicBrainz text search, zero from iTunes). Its + filename, though, says exactly what it is. Pure string work, no network, no + Qt: strips yt-dlp's trailing id (`[1025657891]`, `-WC7gK2kgyTQ` — an 11-char + token is only treated as a YouTube id if it carries a digit, an underscore, + or repeatedly flipping case, or "Underground" would be eaten), drops + `(Official Video)`-style noise, splits `Artist - Title` on a *spaced* hyphen + only (so "Jay-Z" and "350-440-DialTone" survive), collapses a doubled + uploader, reads a leading `1-04`, restores `_s`→`'s`, and falls back to the + iTunes tree (`/Artist/Album/NN Title.ext`), ignoring placeholder dirs + like "Unknown Artist". `fingerprint.candidate_from_filename` wraps it as a + candidate with `source="filename"`, appended to every lookup and standing + alone when there are no matches — the dialog then says where the guess came + from instead of quoting a fabricated confidence. The same filename also + feeds `hint_for(track)`, which ranks real lookup results: token overlap + (words worth double, bare numbers single — "alhambra" identifies a release, + "1961" appears in every compilation spanning it), a coarse **duration** + bucket off the recording lengths AcoustID returns (which is what separates a + 150 s studio take from a 155 s live one), and a year named in the filename + that predates anything the database knows. That last one is not a nicety: + MusicBrainz's own `first-release-date` for "Ahmad Jamal's Alhambra" is + **2002**, because its three original 1961 pressings are in the database + undated — so a second MusicBrainz call would return the same wrong year, and + the file's own name is the only place 1961 exists. + - **`lintunes/mpris.py`** — registers `org.mpris.MediaPlayer2.lintunes` over D-Bus so the desktop's media keys / now-playing popup control playback. Spacebar and arrow keys are handled locally via `MainWindow.eventFilter`. diff --git a/TASKS.md b/TASKS.md index ee38f57..7b4d60a 100644 --- a/TASKS.md +++ b/TASKS.md @@ -9,6 +9,32 @@ When a round closes, move its finished items to `tasks-done.md`. - [ ] archive the done tasks in here to another file, this is crufty.... +## Round 46 (2026-09-02) — The file already told you: done, see tasks-done.md + +Filename-derived proposals when no database knows the song, plus using the +file's own tags/name/duration to rank real lookup results. Left on the table: + +- [ ] **Submit fingerprints back to AcoustID.** The Zuuso case is unfixable by + lookup because nobody ever submitted it — and LinTunes is holding a good + fingerprint plus (after an identify) confirmed tags. `/v2/submit` takes + exactly that. It would make the library better for everyone and fix + trav's own second machine. Needs a user API key (not just the app key) + and must be explicitly opt-in — never submit tags the user hasn't + confirmed. +- [ ] **An LLM pass for the genuinely ambiguous ones.** Raised by trav + 2026-09-02. Worth being precise about where it helps: *not* the Zuuso + case (no data exists to reason over) and *not* the Alhambra year (1961 + isn't in the response either). It helps where there are many plausible + candidates and the signal is semantic — an odd filename convention, or + deciding whether a number is a performance year or a release year. Would + be the app's first LLM dependency; keep it behind the same + "nothing is written until accepted" gate. +- [ ] **A "clean up this filename" action independent of identify.** Half of + what round 46 built is useful with no network at all: 474 files in + Unknown Artist/Unknown Album could have artist/title/track filled from + their names in one pass. Wants the worker+progress split, not a dialog + per track. + ## Round 45 (2026-09-02) — Ask the song what it is: done, see tasks-done.md Right-click → Identify Track…: Chromaprint fingerprint + AcoustID lookup, a diff --git a/lintunes/__init__.py b/lintunes/__init__.py index 4e4d053..32c9129 100644 --- a/lintunes/__init__.py +++ b/lintunes/__init__.py @@ -1,3 +1,3 @@ """LinTunes — iTunes-style music library manager and player for Linux.""" -__version__ = "0.15.0" +__version__ = "0.16.0" diff --git a/lintunes/filename_tags.py b/lintunes/filename_tags.py new file mode 100644 index 0000000..8a66f32 --- /dev/null +++ b/lintunes/filename_tags.py @@ -0,0 +1,164 @@ +"""Reading tags out of a file's own name, for when no database knows the song. + +Fingerprinting answers "who is this?" only for music somebody has already +submitted to AcoustID. A SoundCloud rip of an underground artist isn't in +AcoustID, isn't in MusicBrainz and isn't in iTunes — but its filename usually +says exactly what it is ("B. Clem - Zuuso [1025657891]"), because that's how +yt-dlp and friends name what they download. + +So this module is the offline half of Identify Track: pure string work, no +network, no Qt. It strips the download id and the "(Official Video)" noise, +splits "Artist - Title", and falls back to the iTunes-style folder tree +(/Artist/Album/NN Title.ext) for anything the name itself doesn't say. + +Everything here is a *guess* and is presented as one — the dialog labels these +proposals "from filename" rather than quoting a match confidence. +""" +import re +from pathlib import Path + + +# yt-dlp appends the source's id: "[1025657891]" (SoundCloud, numeric) or +# "-WC7gK2kgyTQ" / "[EtNZnhxWLHo]" (YouTube, 11 chars of base64url). +_YT_ID_CHARS = re.compile(r"^[A-Za-z0-9_-]{11}$") +_NUMERIC_ID = re.compile(r"[\[\(]\s*\d{6,}\s*[\]\)]\s*$|[-_]\d{6,}\s*$") +_BRACKET_ID = re.compile(r"[\[\(]\s*([A-Za-z0-9_-]{11})\s*[\]\)]\s*$") +_TRAILING_ID = re.compile(r"-([A-Za-z0-9_-]{11})\s*$") + +# "(Official Video)", "[HD]", "(mp3)" and friends — noise, never a title. +_NOISE = re.compile( + r"\s*[\(\[]\s*(?:" + # an optional "official", an optional qualifier, then video/audio — + # covers "Official Video", "Official Lyric Video", "Music Video", "Audio" + r"(?:official\s*)?(?:music\s*|lyrics?\s*|full\s*)?(?:video|audio)|" + r"official|lyrics?|visuali[sz]er|" + r"hd|hq|4k|full\s*album|free\s*download|out\s*now|mp3|m4a|flac" + r")\s*[\)\]]", re.I) + +# "02 Title", "1-01 Title", "03. Title" +_LEADING_NUMBER = re.compile(r"^\s*(?:(\d{1,2})[-_ ]+)?(\d{1,3})[\.\)]?\s+(?=\S)") + +# A bare four-digit year. Only trusted when the text holds exactly one, so a +# range ("1958-62", "Tour 1958-1961") is never mistaken for a release year. +_YEAR = re.compile(r"(? bool: + """An 11-char base64url token that reads as random rather than as a word. + + A real word can be 11 characters ("Underground"), so length alone would + eat legitimate titles. Random ids carry a digit, an underscore, or case + that flips repeatedly — an English word does none of those. + """ + if not _YT_ID_CHARS.match(token): + return False + if any(c.isdigit() for c in token) or "_" in token: + return True + transitions = sum( + 1 for a, b in zip(token, token[1:]) + if a.isalpha() and b.isalpha() and a.isupper() != b.isupper()) + return transitions >= 3 + + +def strip_download_id(stem: str) -> str: + """Remove a trailing yt-dlp source id, in brackets or hyphen-attached.""" + text = _NUMERIC_ID.sub("", stem).strip() + match = _BRACKET_ID.search(text) + if match and _looks_like_a_youtube_id(match.group(1)): + return text[:match.start()].strip() + match = _TRAILING_ID.search(text) + if match and _looks_like_a_youtube_id(match.group(1)): + return text[:match.start()].strip() + return text + + +def strip_noise(text: str) -> str: + """Drop "(Official Video)"-style decorations.""" + previous = None + while previous != text: + previous = text + text = _NOISE.sub("", text).strip() + return text.strip(" -_·—–").strip() + + +def restore_apostrophes(text: str) -> str: + return _APOSTROPHE.sub(r"'\1", text) + + +def leading_track_number(text: str) -> tuple[str, int | None, int | None]: + """Split "1-01 Title" into (title, track number, disc number).""" + match = _LEADING_NUMBER.match(text) + if not match: + return text, None, None + disc, number = match.group(1), match.group(2) + rest = text[match.end():].strip() + if not rest: + return text, None, None + return (rest, int(number), int(disc) if disc else None) + + +def year_in(text: str) -> int | None: + """The year a string names, when it names exactly one.""" + years = set(_YEAR.findall(text or "")) + if len(years) != 1: + return None + return int(next(iter(years))) + + +def split_artist_title(text: str) -> tuple[str | None, str]: + """Split "Artist - Title" on the first hyphen that has space around it. + + A hyphen inside a word ("Jay-Z", "re-recorded") is never a separator, and + a doubled artist ("DJ X - DJ X - Track") collapses to one. + """ + parts = [p.strip() for p in re.split(r"\s+[-–—]\s+", text) if p.strip()] + if len(parts) < 2: + return None, text.strip() + # yt-dlp sometimes repeats the uploader: keep one copy. + while len(parts) > 2 and parts[0].casefold() == parts[1].casefold(): + parts.pop(0) + artist = parts[0] + title = " - ".join(parts[1:]) if len(parts) > 2 else parts[1] + return artist, title + + +def parse_filename(location: str) -> dict: + """Best-effort tags from a path. Keys are LinTunes field names; absent + keys mean "the name doesn't say".""" + path = Path(location) + stem = strip_download_id(path.stem) + stem = strip_noise(stem) + stem, number, disc = leading_track_number(stem) + artist, title = split_artist_title(stem) + title = restore_apostrophes(strip_noise(title)).strip() + + fields: dict = {} + if title: + fields["name"] = title + if artist: + fields["artist"] = restore_apostrophes(artist).strip() + if number is not None: + fields["track_number"] = number + if disc is not None: + fields["disc_number"] = disc + + # The iTunes-style tree fills in what the name itself doesn't say: + # /Artist/Album/NN Title.ext + parents = [p.name for p in path.parents] + if len(parents) >= 2: + album_dir, artist_dir = parents[0], parents[1] + if album_dir.casefold() not in _UNKNOWN_DIRS: + fields.setdefault("album", album_dir) + if artist_dir.casefold() not in _UNKNOWN_DIRS: + fields.setdefault("artist", artist_dir) + + year = year_in(path.stem) + if year is not None: + fields["year"] = year + return fields diff --git a/lintunes/fingerprint.py b/lintunes/fingerprint.py index d7566fc..91582fe 100644 --- a/lintunes/fingerprint.py +++ b/lintunes/fingerprint.py @@ -12,13 +12,17 @@ AcoustID application key is user-supplied via Preferences (the Last.fm precedent); registration is free at https://acoustid.org/new-application. """ import json +import re import shutil import subprocess import threading from dataclasses import asdict, dataclass +from pathlib import Path from PyQt6.QtCore import QObject, pyqtSignal +from lintunes import filename_tags + ACOUSTID_LOOKUP_URL = "https://api.acoustid.org/v2/lookup" LOOKUP_META = "recordings releasegroups releases tracks compress" @@ -33,6 +37,10 @@ class IdentifyCandidate: """One proposed identification. Field names match Track/tagging so ``fields()`` can go straight into ``LibraryManager.edit_track_fields``.""" score: float + # "acoustid" (a fingerprint match, so ``score`` is a real confidence) or + # "filename" (a guess read off the file's own name — no confidence to + # quote, and the dialog says so rather than implying one). + source: str = "acoustid" name: str | None = None artist: str | None = None album_artist: str | None = None @@ -45,7 +53,7 @@ class IdentifyCandidate: def fields(self) -> dict: """Proposed edits keyed by LinTunes field names, Nones omitted.""" return {k: v for k, v in asdict(self).items() - if k != "score" and v is not None} + if k not in ("score", "source") and v is not None} # ---- pure parsing/ranking helpers (offline-testable) ---- @@ -78,18 +86,75 @@ def _oldest_year(releasegroup: dict) -> int | None: return min(years) if years else None +# Secondary types that mean "this is not the album the song came from". +# Live is deliberately absent: a live album is a real album, and demoting it +# buried "Ahmad Jamal's Alhambra" under a compilation for a file whose own +# name said "Live At The Alhambra". +_DEMOTED_SECONDARY = {"compilation", "interview", "remix", "dj-mix", + "mixtape/street", "demo", "audiobook"} + + def _releasegroup_sort_key(releasegroup: dict) -> tuple: - """Rank a release group for the default proposal: the earliest proper - studio Album first, then EPs/Singles, then Compilations/Live/etc.""" + """Rank a release group for the default proposal: a proper album first, + then EPs/Singles, with compilations and the like pushed down.""" rg_type = releasegroup.get("type") or "" secondary = releasegroup.get("secondarytypes") or [] - if not secondary: - rank = {"Album": 0, "EP": 1, "Single": 2}.get(rg_type, 4) - else: - rank = 3 if rg_type == "Album" else 4 + rank = {"Album": 0, "EP": 1, "Single": 2}.get(rg_type, 4) + if any((s or "").casefold() in _DEMOTED_SECONDARY for s in secondary): + rank += 3 return (rank, _oldest_year(releasegroup) or 9999) +_TOKEN = re.compile(r"[^\W\d_]+|\d+", re.UNICODE) +_STOPWORDS = {"the", "a", "an", "of", "and", "in", "at", "on", "live", + "feat", "ft", "remastered", "version", "mix", "edit"} + + +def _tokens(text: str) -> set: + """Comparable words from a title/album/filename.""" + return {t.casefold() for t in _TOKEN.findall(text or "") + if len(t) > 1 and t.casefold() not in _STOPWORDS} + + +def _duration_bucket(recording: dict, duration: float | None) -> int: + """How well a recording's length matches the file's, coarsely. + + AcoustID often returns several recordings of the same song — a studio + take, a live take, an edit — and their lengths are what tell them apart. + Buckets rather than raw difference so a second or two never outweighs the + other evidence, but a clearly different take sinks. + """ + other = recording.get("duration") + if duration is None or not isinstance(other, (int, float)): + return 1 # unknown: behind an exact match, ahead of a bad one + diff = abs(other - duration) + if diff <= 1.5: + return 0 + if diff <= 4: + return 1 + if diff <= 12: + return 2 + return 3 + + +def _hint_overlap(hint_tokens: set, candidate: IdentifyCandidate) -> int: + """How much a candidate echoes what the file already claims to be. + + The distinctive words in "Snowfall (Live At The Alhambra_1961)" — + *snowfall*, *alhambra*, *1961* — are the best evidence available about + which of eight plausible releases this actually is, and ignoring them was + why a compilation could outrank the album named in the filename. + """ + if not hint_tokens: + return 0 + text = " ".join(p for p in (candidate.name, candidate.artist, + candidate.album_artist, candidate.album) if p) + # A word like "alhambra" names a release; a bare number like "1961" turns + # up in every compilation spanning that year, so it counts for less. + return sum(2 if not token.isdigit() else 1 + for token in hint_tokens & _tokens(text)) + + def _pick_release(releasegroup: dict) -> dict | None: """Earliest dated release that carries mediums (track positions); falls back to the earliest dated one, then the first.""" @@ -126,7 +191,8 @@ def _track_position(release: dict | None) -> tuple[int | None, int | None, def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, - limit: int = MAX_CANDIDATES) -> list[IdentifyCandidate]: + limit: int = MAX_CANDIDATES, hint: str = "", + duration: float | None = None) -> list[IdentifyCandidate]: """Turn an AcoustID lookup response into ranked candidates. One candidate per release group of each matched recording, so the @@ -135,16 +201,23 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, the minimum across all its release groups' releases — even when the proposed album is a later compilation: years sort by when the song first came out, not by which pressing this file happens to be from. + + ``hint`` is whatever the file already claims to be (its tags and its + filename). It never invents a candidate, but it breaks ties: a release + whose words echo the hint outranks one that doesn't, and a year named in + the hint wins when it predates anything the database knows — MusicBrainz + often holds only a reissue's date, with the original pressing undated. """ - candidates = [] + hint_tokens = _tokens(hint) + hint_year = _single_year(hint) + scored = [] seen = set() - results = sorted(payload.get("results", []), - key=lambda r: r.get("score", 0), reverse=True) - for result in results: + for result in payload.get("results", []): score = result.get("score", 0) if score < threshold: continue for recording in result.get("recordings", []): + length = _duration_bucket(recording, duration) title = recording.get("title") artist = _join_artists(recording.get("artists")) groups = sorted(recording.get("releasegroups") or [], @@ -152,10 +225,16 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, original_years = [y for rg in groups if (y := _oldest_year(rg)) is not None] original_year = min(original_years) if original_years else None + if hint_year is not None and (original_year is None + or hint_year < original_year): + original_year = hint_year if not groups: if title: - candidates.append(IdentifyCandidate( - score=score, name=title, artist=artist)) + bare = IdentifyCandidate(score=score, name=title, + artist=artist, year=hint_year) + scored.append(((-round(score, 2), length, + -_hint_overlap(hint_tokens, bare), + 4, 9999), bare)) continue for releasegroup in groups: album = releasegroup.get("title") @@ -165,7 +244,7 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, seen.add(key) number, count, disc = _track_position( _pick_release(releasegroup)) - candidates.append(IdentifyCandidate( + candidate = IdentifyCandidate( score=score, name=title, artist=artist, @@ -175,8 +254,47 @@ def parse_lookup(payload: dict, threshold: float = SCORE_THRESHOLD, track_number=number, track_count=count, disc_number=disc, - )) - return candidates[:limit] + ) + rank, rg_year = _releasegroup_sort_key(releasegroup) + scored.append(((-round(score, 2), length, + -_hint_overlap(hint_tokens, candidate), + rank, rg_year), candidate)) + scored.sort(key=lambda pair: pair[0]) + return [candidate for _key, candidate in scored][:limit] + + +def _single_year(text: str) -> int | None: + """The year a string names, when it names exactly one (so a range like + "1958-62" is never mistaken for a release year).""" + return filename_tags.year_in(text or "") + + +def hint_for(track) -> str: + """What a track already claims to be — its tags plus its filename. Used + to rank lookup results, and to fall back on when nothing matches.""" + parts = [track.name or "", track.artist or "", track.album or ""] + if track.location: + parts.append(Path(track.location).stem) + return " ".join(p for p in parts if p) + + +def candidate_from_filename(track) -> IdentifyCandidate | None: + """A proposal read off the file's own name, for the very common case that + no database has ever heard of the song. + + Underground and self-released music simply isn't in AcoustID — the + fingerprint is fine, there is just nothing to match it against — but the + download named the file "Artist - Title [id]", which is how a human reads + it too. Returns None when the name yields nothing usable. + """ + if not track.location: + return None + fields = filename_tags.parse_filename(track.location) + candidate = IdentifyCandidate(score=0.0, source="filename") + for key, value in fields.items(): + if hasattr(candidate, key): + setattr(candidate, key, value) + return candidate if candidate.fields() else None # ---- subprocess + network ---- @@ -209,8 +327,8 @@ def fingerprint_file(location: str) -> tuple[int, str]: raise OSError(f"couldn't read fpcalc output for {name}") from e -def lookup_fingerprint(api_key: str, duration: int, - fingerprint: str) -> list[IdentifyCandidate]: +def lookup_fingerprint(api_key: str, duration: int, fingerprint: str, + hint: str = "") -> list[IdentifyCandidate]: import requests response = requests.post( ACOUSTID_LOOKUP_URL, @@ -224,25 +342,41 @@ def lookup_fingerprint(api_key: str, duration: int, if payload.get("status") != "ok": message = (payload.get("error") or {}).get("message", "lookup failed") raise RuntimeError(message) - return parse_lookup(payload) + return parse_lookup(payload, hint=hint, duration=duration) class TrackIdentifier(QObject): """One fingerprint+lookup running off the GUI thread. ``finished`` carries {"track_id": int, "candidates": [IdentifyCandidate]} - or {"track_id": int, "error": str}. + or {"track_id": int, "error": str}. A filename guess is appended to the + fingerprint's matches, and stands alone when there are none — a lookup + that found nothing is not the same as nothing to propose. """ finished = pyqtSignal(object) - def identify(self, track_id: int, location: str, api_key: str): + def identify(self, track, api_key: str): + track_id = track.track_id + location = track.location + hint = hint_for(track) + guess = candidate_from_filename(track) + def work(): try: duration, fp = fingerprint_file(location) - candidates = lookup_fingerprint(api_key, duration, fp) - self.finished.emit( - {"track_id": track_id, "candidates": candidates}) + candidates = lookup_fingerprint(api_key, duration, fp, hint) except Exception as e: - self.finished.emit({"track_id": track_id, "error": str(e)}) + # A file too short or too odd to fingerprint still has a name. + if guess is not None: + self.finished.emit({"track_id": track_id, + "candidates": [guess], + "note": str(e)}) + else: + self.finished.emit({"track_id": track_id, "error": str(e)}) + return + if guess is not None: + candidates = candidates + [guess] + self.finished.emit( + {"track_id": track_id, "candidates": candidates}) threading.Thread(target=work, daemon=True).start() diff --git a/lintunes/gui/identify_dialog.py b/lintunes/gui/identify_dialog.py index f53afd3..1e11ae6 100644 --- a/lintunes/gui/identify_dialog.py +++ b/lintunes/gui/identify_dialog.py @@ -35,19 +35,20 @@ class IdentifyDialog(QDialog): self.cancel_all = False layout = QVBoxLayout(self) - best = candidates[0] - header = QLabel(f"Identified “{track.name or '(untitled)'}” — " - f"match confidence {round(best.score * 100)}%") - header.setWordWrap(True) - layout.addWidget(header) + self._header = QLabel() + self._header.setWordWrap(True) + layout.addWidget(self._header) # All alternates at a glance — a candidate summarizes to one line, so # a dropdown beats AlbumArtDialog's blind Next-clicking. self._combo = QComboBox() for c in candidates: year = f" ({c.year})" if c.year else "" - self._combo.addItem(" — ".join( - p for p in (c.name, c.artist, c.album) if p) + year) + summary = " — ".join( + p for p in (c.name, c.artist, c.album) if p) + year + if c.source == "filename": + summary = f"From the filename: {summary}" + self._combo.addItem(summary) layout.addWidget(self._combo) grid = QGridLayout() @@ -105,6 +106,17 @@ class IdentifyDialog(QDialog): iff the candidate proposes a value that differs from the current one; rows with nothing proposed are unchecked and disabled.""" candidate = self._candidates[index] + name = self._track.name or "(untitled)" + if candidate.source == "filename": + # No database matched this, so there is no confidence to quote — + # say where the proposal actually came from instead of implying + # something recognized the audio. + self._header.setText( + f"No database knows “{name}”. Read from the file's name:") + else: + self._header.setText( + f"Identified “{name}” — match confidence " + f"{round(candidate.score * 100)}%") for field, _label, is_number, _max in _FIELDS: current = self._current_value(field) proposed = getattr(candidate, field) diff --git a/lintunes/gui/main_window.py b/lintunes/gui/main_window.py index 882e76e..7e3e563 100644 --- a/lintunes/gui/main_window.py +++ b/lintunes/gui/main_window.py @@ -1138,7 +1138,7 @@ class MainWindow(QMainWindow): self._identifier = TrackIdentifier() self._identifier.finished.connect( lambda result, t=track, k=api_key: self._on_identified(t, k, result)) - self._identifier.identify(track.track_id, track.location, api_key) + self._identifier.identify(track, api_key) def _on_identified(self, track, api_key: str, result: dict): if "error" in result: @@ -1146,7 +1146,8 @@ class MainWindow(QMainWindow): f"Couldn't identify “{track.name}”: {result['error']}", 6000) elif not result["candidates"]: self.statusBar().showMessage( - f"No match found for “{track.name}”", 6000) + f"Nothing to propose for “{track.name}” — no database match " + f"and the filename says nothing", 6000) else: self.statusBar().clearMessage() dialog = IdentifyDialog(track, result["candidates"], diff --git a/tasks-done.md b/tasks-done.md index 54f5698..3be726c 100644 --- a/tasks-done.md +++ b/tasks-done.md @@ -1,5 +1,51 @@ ## Done +### Round 46 (2026-09-02) — The file already told you (v0.16.0) + +Round 45 shipped and trav broke it in two places within the hour, both real. +"B. Clem - Zuuso [1025657891]" fingerprinted fine and matched **nothing**, and +"Snowfall (Live At The Alhambra_1961)" matched but proposed a 2002 reissue year +and ranked a compilation above the album its own filename named. + +- [x] **The limit is the database, not the fingerprint.** Verified before + building anything: that Zuuso fingerprint returns zero AcoustID results, + a MusicBrainz text search for "Zuuso" returns zero, and iTunes returns + zero (and irrelevant fuzz for "B. Clem"). The track exists in no metadata + service on earth. So no smarter lookup — LLM or otherwise — could have + answered it, because there was nothing to answer with. +- [x] **But the filename knew.** New `lintunes/filename_tags.py` reads tags out + of the name yt-dlp gave the file. "B. Clem - Zuuso [1025657891]" → artist + "B. Clem", name "Zuuso", which is exactly what trav said it was. Checked + against all 474 files in the untagged folder, not against invented + examples. +- [x] **Careful about what it strips.** An 11-character trailing token is only + a YouTube id if it carries a digit, an underscore, or case that flips + repeatedly — otherwise "Cold Draft-Underground" loses a word. A hyphen + only splits artist from title when it has spaces around it, so "Jay-Z" + and "350-440-DialTone" survive. A 4-digit year is never a numeric id. +- [x] **A guess is labelled a guess.** `IdentifyCandidate.source` is + "acoustid" or "filename"; the dialog quotes a match confidence only for + the former and says "No database knows this — read from the file's name" + for the latter, rather than inventing a percentage. +- [x] **The file now ranks the results too.** `hint_for(track)` feeds tag + + filename tokens into `parse_lookup`: overlap scoring (words count double, + bare numbers single — "alhambra" identifies a release, "1961" appears in + every compilation spanning it), a coarse duration bucket against the + recording lengths AcoustID returns, and a filename year that predates + what the database knows. Alhambra now ranks the right album first with + year 1961 instead of a compilation with 2002. +- [x] **Live is no longer a demerit.** It was lumped in with Compilation, which + buried a live album for a file whose name said "Live At The Alhambra". + Only compilations, remixes, interviews and the like are demoted now. +- [x] **Why the year needed the filename at all:** MusicBrainz's own + `first-release-date` for "Ahmad Jamal's Alhambra" is *2002* — its three + original 1961 pressings are in the database undated. A second MusicBrainz + call, the obvious fix, returns the same wrong answer. 1961 exists only in + trav's filename. +- [x] Not done, deliberately: trav rejected "never propose a shorter title" — + truncating is *wanted*, because the garbage in a title is usually the + part being dropped. + ### Round 45 (2026-09-02) — Ask the song what it is (v0.15.0) Some files arrive with the title right and everything else missing or wrong, diff --git a/tests/test_round46.py b/tests/test_round46.py new file mode 100644 index 0000000..c5175ed --- /dev/null +++ b/tests/test_round46.py @@ -0,0 +1,227 @@ +"""Round 46 — Identify Track stops giving up when no database knows the song. + +Two real failures from trav's library drove this. "B. Clem - Zuuso +[1025657891]" fingerprinted fine and returned *zero* AcoustID results — the +track is underground, so it is in no database at all, and no amount of +cleverness can look up what does not exist. But its filename said exactly what +it was, which is how trav knew. And "Snowfall (Live At The Alhambra_1961)" +matched, yet proposed a 2002 reissue year and ranked a compilation above the +album the filename named, because nothing compared the results against what +the file already said. + +So the file itself is now evidence: it seeds a proposal when the lookup finds +nothing, and it ranks the lookup's results when it finds several. +""" +import pytest + +from lintunes.fingerprint import ( + IdentifyCandidate, candidate_from_filename, hint_for, parse_lookup, +) +from lintunes.filename_tags import ( + leading_track_number, parse_filename, split_artist_title, + strip_download_id, strip_noise, year_in, +) + + +class _Track: + def __init__(self, name="", artist="", album="", location=""): + self.track_id = 1 + self.name = name + self.artist = artist + self.album = album + self.location = location + + +UNKNOWN = "/m/Unknown Artist/Unknown Album/" + + +class TestStrippingDownloadIds: + @pytest.mark.parametrize("stem, expected", [ + ("B. Clem - Zuuso [1025657891]", "B. Clem - Zuuso"), + ("Stardust-321895497", "Stardust"), + ("Cold Draft-WC7gK2kgyTQ", "Cold Draft"), + ("Self Esteem [EtNZnhxWLHo]", "Self Esteem"), + ("Caught Up in the Rapture [Oz-b86LZ21c]", "Caught Up in the Rapture"), + ]) + def test_yt_dlp_ids_are_removed(self, stem, expected): + assert strip_download_id(stem) == expected + + def test_a_real_word_is_not_mistaken_for_a_youtube_id(self): + """"Underground" is 11 characters too — length alone can't decide.""" + assert strip_download_id("Cold Draft-Underground") == \ + "Cold Draft-Underground" + + def test_a_year_is_not_mistaken_for_a_numeric_id(self): + assert strip_download_id("Cross Country Tour 1958-1961") == \ + "Cross Country Tour 1958-1961" + + +class TestStrippingNoise: + @pytest.mark.parametrize("text", [ + "Dance II (Official Video)", + "Dance II (Official Music Video)", + "Dance II (Official Lyric Video)", + "Dance II [HD]", + "Dance II (Audio)", + ]) + def test_decorations_go(self, text): + assert strip_noise(text) == "Dance II" + + def test_a_real_parenthetical_stays(self): + assert strip_noise("Take Off (ASOT 1159)") == "Take Off (ASOT 1159)" + + +class TestSplittingAndNumbering: + def test_artist_title_split(self): + assert split_artist_title("B. Clem - Zuuso") == ("B. Clem", "Zuuso") + + def test_hyphen_inside_a_word_is_not_a_separator(self): + assert split_artist_title("350-440-DialTone") == \ + (None, "350-440-DialTone") + + def test_a_repeated_uploader_collapses(self): + assert split_artist_title("Sean Dream - Sean Dream - tomorrow111") == \ + ("Sean Dream", "tomorrow111") + + def test_leading_track_and_disc_numbers(self): + assert leading_track_number("1-04 Sitting Back") == \ + ("Sitting Back", 4, 1) + assert leading_track_number("02 nashville cats") == \ + ("nashville cats", 2, None) + + def test_a_number_that_is_the_whole_title_is_kept(self): + assert leading_track_number("1979") == ("1979", None, None) + + +class TestYearMining: + def test_a_single_year_is_read(self): + assert year_in("Snowfall (Live At The Alhambra_1961)") == 1961 + + def test_a_range_is_refused(self): + """"1958-62" and "1958-1961" name a span, not a release year.""" + assert year_in("Cross Country Tour 1958-1961") is None + + def test_no_year_is_none(self): + assert year_in("Zuuso") is None + + +class TestFilenameProposal: + def test_the_zuuso_case(self): + """The track that started this: fingerprinted fine, matched nothing, + and its name said exactly what trav already knew it was.""" + track = _Track(name="B. Clem - Zuuso [1025657891]", + location=UNKNOWN + "B. Clem - Zuuso [1025657891].mp3") + candidate = candidate_from_filename(track) + assert candidate.source == "filename" + assert candidate.fields() == {"name": "Zuuso", "artist": "B. Clem"} + + def test_apostrophes_come_back(self): + track = _Track(location=UNKNOWN + "cehryl - That_s So Sick! [738273660].mp3") + assert candidate_from_filename(track).name == "That's So Sick!" + + def test_the_folder_tree_fills_what_the_name_omits(self): + track = _Track(location="/m/Onra/Long Distance/1-04 Sitting Back.mp3") + fields = candidate_from_filename(track).fields() + assert fields["artist"] == "Onra" + assert fields["album"] == "Long Distance" + assert fields["name"] == "Sitting Back" + assert fields["track_number"] == 4 + + def test_placeholder_folders_are_not_treated_as_tags(self): + track = _Track(location=UNKNOWN + "blurf.mp3") + fields = candidate_from_filename(track).fields() + assert "artist" not in fields and "album" not in fields + + def test_a_track_with_no_file_proposes_nothing(self): + assert candidate_from_filename(_Track()) is None + + +def _payload(*recordings, score=0.9): + return {"status": "ok", + "results": [{"score": score, "recordings": list(recordings)}]} + + +def _recording(title, duration=None, groups=()): + rec = {"title": title, "artists": [{"name": "The Band"}], + "releasegroups": list(groups)} + if duration is not None: + rec["duration"] = duration + return rec + + +def _group(title, year, type_="Album", secondary=None): + return {"title": title, "type": type_, + "secondarytypes": list(secondary or []), + "releases": [{"date": {"year": year} if year else {}}]} + + +class TestRankingAgainstTheFile: + def test_the_album_the_filename_names_wins(self): + payload = _payload(_recording("Snowfall", groups=[ + _group("Some Other Record", 1970), + _group("Ahmad Jamal's Alhambra", 2002, secondary=["Live"]), + ])) + hint = "Snowfall (Live At The Alhambra_1961)" + assert parse_lookup(payload, hint=hint)[0].album == \ + "Ahmad Jamal's Alhambra" + + def test_a_live_album_is_no_longer_demoted(self): + """Live is what a live album *is*; only compilations get pushed down.""" + payload = _payload(_recording("Snowfall", groups=[ + _group("Greatest Hits", 1990, secondary=["Compilation"]), + _group("The Live Album", 1995, secondary=["Live"]), + ])) + assert parse_lookup(payload)[0].album == "The Live Album" + + def test_duration_picks_the_right_take(self): + """Two recordings of one song; only the length tells them apart.""" + payload = _payload( + _recording("Snowfall", duration=155.0, + groups=[_group("The Long Take", 2000)]), + _recording("Snowfall", duration=150.3, + groups=[_group("The Short Take", 2000)]), + ) + assert parse_lookup(payload, duration=150)[0].album == "The Short Take" + + def test_a_year_in_the_filename_beats_a_reissue_date(self): + """MusicBrainz often has only the CD reissue dated, with the original + pressing undated — so the file's own 1961 is the better evidence.""" + payload = _payload(_recording("Snowfall", groups=[ + _group("Ahmad Jamal's Alhambra", 2002)])) + hint = "Snowfall (Live At The Alhambra_1961)" + assert parse_lookup(payload, hint=hint)[0].year == 1961 + + def test_a_later_filename_year_never_overrides_an_earlier_one(self): + payload = _payload(_recording("Song", groups=[_group("Album", 1967)])) + assert parse_lookup(payload, hint="Song (remaster 1999)")[0].year == 1967 + + def test_hint_never_invents_a_candidate(self): + assert parse_lookup({"status": "ok", "results": []}, + hint="anything at all") == [] + + +class TestHint: + def test_hint_gathers_tags_and_filename(self): + track = _Track(name="Snowfall", artist="Ahmad Jamal", + location=UNKNOWN + "Snowfall (Live At The Alhambra_1961).mp3") + hint = hint_for(track) + assert "Snowfall" in hint and "Ahmad Jamal" in hint and "1961" in hint + + +class TestDialogLabelsAGuessHonestly: + def test_filename_candidates_say_where_they_came_from(self, qapp): + from lintunes.gui.identify_dialog import IdentifyDialog + track = _Track(name="B. Clem - Zuuso [1025657891]") + guess = IdentifyCandidate(score=0.0, source="filename", + name="Zuuso", artist="B. Clem") + dialog = IdentifyDialog(track, [guess]) + assert "file's name" in dialog._header.text().lower() + # No fabricated confidence percentage for something nothing matched. + assert "%" not in dialog._header.text() + assert "From the filename" in dialog._combo.itemText(0) + + def test_acoustid_candidates_still_quote_confidence(self, qapp): + from lintunes.gui.identify_dialog import IdentifyDialog + candidate = IdentifyCandidate(score=0.98, name="Snowfall") + dialog = IdentifyDialog(_Track(name="x"), [candidate]) + assert "98%" in dialog._header.text()