Files
lintunes/lintunes/art_search.py
T
travandClaude Opus 5 3150df0f89 v0.23.0: covers for the whole album
- Multi-select Get Info keeps the art square; a pasted cover goes into
  every selected track (shared cover shown, else "mixed artwork").
- Paste Artwork button + Ctrl+V anywhere outside a text field; a caption
  says what happened. Clipboard reads that don't decode are retried and
  never staged; every attempt is logged at INFO.
- Album art search queries Deezer alongside iTunes and ranks by album
  match (iTunes has no copy of Digable Planets' Reachin' at all).
- embed_artwork shared by Get Info and Download Album Art; single-track
  Get Info now invalidates the MPRIS art cache.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-22 09:38:39 -07:00

237 lines
8.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Album-art lookup via the iTunes Search API and Deezer (neither needs a key).
Two catalogs because neither is complete: iTunes has no copy at all of some
well-known albums (Digable Planets' *Reachin'* returns nothing however it's
spelled) that Deezer finds on the first try, and vice versa. Results from both
are merged and ranked by how well they match what was asked for, so a stray
hit from the artist's other album never outranks the real one.
Pure parsing/ranking helpers are separated from the network calls so they can
be tested offline; ``AlbumArtFetcher`` runs the whole search+download on a
daemon thread (the lastfm pattern) and reports back over a Qt signal, which
is delivered queued on the GUI thread.
"""
import re
import threading
import unicodedata
from dataclasses import dataclass
from PyQt6.QtCore import QObject, pyqtSignal
ITUNES_SEARCH_URL = "https://itunes.apple.com/search"
DEEZER_SEARCH_URL = "https://api.deezer.com/search/album"
ART_SIZE = 600 # px; the API hands out 100x100 URLs that scale on request
TIMEOUT_S = 15
@dataclass
class ArtCandidate:
artist: str
album: str
art_url: str # already upgraded to ART_SIZE
source: str = "iTunes"
def upgrade_artwork_url(url: str, size: int = ART_SIZE) -> str:
"""The API returns .../100x100bb.jpg thumbnails; Apple's image server
serves the same asset at (almost) any requested size."""
return url.replace("100x100", f"{size}x{size}")
def parse_results(payload: dict) -> list[ArtCandidate]:
candidates = []
for item in payload.get("results", []):
url = item.get("artworkUrl100")
if not url:
continue
candidates.append(ArtCandidate(
artist=item.get("artistName", ""),
album=item.get("collectionName", ""),
art_url=upgrade_artwork_url(url),
source="iTunes",
))
return candidates
def parse_deezer_results(payload: dict) -> list[ArtCandidate]:
candidates = []
for item in payload.get("data", []) or []:
url = (item.get("cover_xl") or item.get("cover_big")
or item.get("cover_medium"))
if not url:
continue
candidates.append(ArtCandidate(
artist=(item.get("artist") or {}).get("name", ""),
album=item.get("title", ""),
art_url=url,
source="Deezer",
))
return candidates
_EDITION_NOISE = re.compile(
r"\b(deluxe|remaster(ed)?|expanded|anniversary|special|bonus track"
r"|collector'?s)\b.*$", re.IGNORECASE)
_TRAILING_GROUP = re.compile(r"\s*[\(\[][^\(\)\[\]]*[\)\]]\s*$")
def _plain_quotes(text: str) -> str:
return (text.replace("’", "'").replace("‘", "'")
.replace("“", '"').replace("”", '"'))
def _simplify_album(album: str) -> str:
"""The title without its trailing ``(...)``/``[...]`` groups or edition
noise: "Reachin' (A New Refutation of Time and Space)" -> "Reachin'"."""
text = _plain_quotes(album).strip()
while True:
stripped = _TRAILING_GROUP.sub("", text)
if stripped == text or not stripped:
break
text = stripped
text = _EDITION_NOISE.sub("", text).strip(" -–—:,")
return text or _plain_quotes(album).strip()
def _normalize(text: str) -> str:
"""Case-, accent- and punctuation-blind form used to compare titles."""
text = unicodedata.normalize("NFKD", _plain_quotes(text))
text = "".join(c for c in text if not unicodedata.combining(c))
return " ".join(re.sub(r"[^\w\s]", " ", text.casefold()).split())
def rank_candidates(candidates: list[ArtCandidate], artist: str,
album: str) -> list[ArtCandidate]:
"""Best match first; duplicates (same source, same normalized artist and
album) dropped. The album decides more than the artist: a right-artist
wrong-album hit is exactly the result that must not win."""
want_album = _normalize(album)
want_simple = _normalize(_simplify_album(album))
want_artist = _normalize(artist)
sources = {"iTunes": 0, "Deezer": 1}
def key(indexed):
index, c = indexed
got = _normalize(c.album)
if got == want_album:
album_score = 0
elif _normalize(_simplify_album(c.album)) == want_simple:
album_score = 1
elif want_simple and want_simple in got:
album_score = 2
else:
album_score = 3
got_artist = _normalize(c.artist)
artist_score = 0 if got_artist == want_artist else (
1 if want_artist and want_artist in got_artist else 2)
return (album_score, artist_score, sources.get(c.source, 9), index)
seen = set()
unique = []
for c in candidates:
ident = (c.source, _normalize(c.artist), _normalize(c.album))
if ident in seen:
continue
seen.add(ident)
unique.append(c)
return [c for _i, c in sorted(enumerate(unique), key=key)]
def _itunes_query(term: str, limit: int) -> list[ArtCandidate]:
import requests
response = requests.get(
ITUNES_SEARCH_URL,
params={"term": term, "entity": "album", "media": "music",
"limit": limit},
timeout=TIMEOUT_S,
)
response.raise_for_status()
return parse_results(response.json())
def search_itunes(artist: str, album: str, limit: int = 5) -> list[ArtCandidate]:
found = _itunes_query(f"{artist} {album}".strip(), limit)
simple = _simplify_album(album)
if not found and simple != album.strip():
found = _itunes_query(f"{artist} {simple}".strip(), limit)
return found
def _deezer_query(query: str, limit: int) -> list[ArtCandidate]:
import requests
response = requests.get(DEEZER_SEARCH_URL,
params={"q": query, "limit": limit},
timeout=TIMEOUT_S)
response.raise_for_status()
payload = response.json()
if "error" in payload:
raise RuntimeError(payload["error"].get("message", "Deezer error"))
return parse_deezer_results(payload)
def search_deezer(artist: str, album: str, limit: int = 5) -> list[ArtCandidate]:
simple = _simplify_album(album).replace('"', "")
strict = f'album:"{simple}"'
if artist:
strict = f'artist:"{artist.replace(chr(34), "")}" ' + strict
found = _deezer_query(strict, limit)
if not found:
found = _deezer_query(f"{artist} {simple}".strip(), limit)
return found
def search_album_art(artist: str, album: str, limit: int = 5) -> list[ArtCandidate]:
"""Both catalogs, merged and ranked. One source failing is not an error
while the other answers; both failing raises the first failure."""
found: list[ArtCandidate] = []
errors: list[Exception] = []
for search in (search_itunes, search_deezer):
try:
found.extend(search(artist, album, limit))
except Exception as e: # network, HTTP, JSON — the other may still work
errors.append(e)
if errors and len(errors) == 2:
raise errors[0]
return rank_candidates(found, artist, album)
def fetch_image(url: str) -> tuple[bytes, str]:
"""Download an image; returns (bytes, mime type)."""
import requests
response = requests.get(url, timeout=TIMEOUT_S)
response.raise_for_status()
mime = response.headers.get("Content-Type", "image/jpeg").split(";")[0]
return response.content, mime or "image/jpeg"
class AlbumArtFetcher(QObject):
"""One search-or-download running off the GUI thread.
``search_finished`` carries {"candidates": [ArtCandidate, ...]} or
{"error": str}; ``image_finished`` carries {"candidate": ArtCandidate,
"image": bytes, "mime": str} or {"error": str}.
"""
search_finished = pyqtSignal(object)
image_finished = pyqtSignal(object)
def search(self, artist: str, album: str):
def work():
try:
candidates = search_album_art(artist, album)
self.search_finished.emit({"candidates": candidates})
except Exception as e:
self.search_finished.emit({"error": str(e)})
threading.Thread(target=work, daemon=True).start()
def fetch(self, candidate: ArtCandidate):
def work():
try:
data, mime = fetch_image(candidate.art_url)
self.image_finished.emit(
{"candidate": candidate, "image": data, "mime": mime})
except Exception as e:
self.image_finished.emit({"error": str(e)})
threading.Thread(target=work, daemon=True).start()