diff --git a/CLAUDE.md b/CLAUDE.md index f277df5..7e1192b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -365,6 +365,19 @@ persistence) → GUI (Qt widgets that read the manager and connect to its signal get `gui/url_import_progress.py`, a non-modal per-song list (Hide keeps downloading; clicking the status-bar line reopens it). `LTSTART` carries the id, and `ERROR: [site] : …` lines become `item_failed`. + **A Spotify link** (Round 76, `spotify_link.py`, ported from trav's + `spotify-youtube` script) can't be downloaded, but Spotify's public + embed page (`open.spotify.com/embed//`, a `__NEXT_DATA__` + JSON block, no key) names every song with its **length**. `SpotifyProbe` + feeds those to the same checklist; `SpotifyImportWorker` (a + `UrlImportWorker` with the same signals, rows keyed by Spotify id) runs a + `ytsearch5:` listing per song, takes the result closest to Spotify's + length (`rank_candidates`: within 10 s, then no "live"/"cover" the + Spotify title doesn't have — the shazam-import lesson), falls back to the + next result if a download fails, and writes Spotify's title/artist/album/ + track number into the mp3 **before** emitting `downloaded`, so the + import files it under the right artist and Identify ranks by that artist. + If the embed page changes shape, `parse_embed` raises `SpotifyError`. - **`lintunes/mpris.py`** — registers `org.mpris.MediaPlayer2.lintunes` over D-Bus so the desktop's media keys / now-playing popup control playback. Spacebar and diff --git a/lintunes/__init__.py b/lintunes/__init__.py index c55d6d9..0a27803 100644 --- a/lintunes/__init__.py +++ b/lintunes/__init__.py @@ -1,3 +1,3 @@ """LinTunes — iTunes-style music library manager and player for Linux.""" -__version__ = "0.36.0" +__version__ = "0.37.0" diff --git a/lintunes/gui/main_window.py b/lintunes/gui/main_window.py index 17fbfc4..2f98348 100644 --- a/lintunes/gui/main_window.py +++ b/lintunes/gui/main_window.py @@ -1340,11 +1340,18 @@ class MainWindow(QMainWindow): self._url_note = (f"tags proposed from filenames only ({problem})" if problem else "") - # The chosen songs by their own pages, never the list link: that - # would download the whole list again, and a Mix reshuffles. - worker = url_import.UrlImportWorker( - dialog.download_urls(), self, - cookies_browser=cookies) + spotify = dialog.spotify_tracks() + if spotify: + # Spotify won't hand over audio: each song is found on YouTube + # and arrives already wearing Spotify's tags. + worker = url_import.SpotifyImportWorker( + spotify, self, cookies_browser=cookies) + else: + # The chosen songs by their own pages, never the list link: that + # would download the whole list again, and a Mix reshuffles. + worker = url_import.UrlImportWorker( + dialog.download_urls(), self, + cookies_browser=cookies) if self._url_progress is not None: self._url_progress.close() self._url_progress.deleteLater() diff --git a/lintunes/gui/url_import_dialog.py b/lintunes/gui/url_import_dialog.py index eb98482..ecf6276 100644 --- a/lintunes/gui/url_import_dialog.py +++ b/lintunes/gui/url_import_dialog.py @@ -18,6 +18,11 @@ first row and returned 2,967 rows in which the linked song appeared 8 times (the Mix loops back on itself). So rows are de-duplicated by id, and a song-in-a-list link gets **Just This Song**, which works from the first moment and downloads the song's own page with the list stripped off. + +A Spotify link (Round 76) is listed the same way, from Spotify's own page +(``url_import.SpotifyProbe``): a track goes straight through, an album or +playlist is a checklist with everything ticked. The caller reads the chosen +songs from ``spotify_tracks()`` and finds each one on YouTube. """ from urllib.parse import parse_qs, urlsplit @@ -28,7 +33,9 @@ from PyQt6.QtWidgets import ( QVBoxLayout, ) -from lintunes.url_import import PlaylistProbe, link_kind, looks_like_url +from lintunes.url_import import ( + PlaylistProbe, SpotifyProbe, link_kind, looks_like_url, +) ENTRY_ROLE = Qt.ItemDataRole.UserRole @@ -72,7 +79,7 @@ class UrlImportDialog(QDialog): layout = QVBoxLayout(self) layout.addWidget(QLabel("Paste a link to a song, or to a playlist of " - "songs:")) + "songs (YouTube, Spotify, …):")) self._url = QLineEdit() self._url.setPlaceholderText("https://…") clipboard = QApplication.clipboard().text().strip() @@ -158,6 +165,15 @@ class UrlImportDialog(QDialog): chosen = self.chosen_entries() return [e["url"] for e in chosen] if chosen else [self.url()] + def spotify_tracks(self) -> list: + """The chosen songs of a Spotify link, as ``SpotifyTrack``s (empty + for any other link): the ticked ones, or the link's one song.""" + if self._kind != "spotify": + return [] + entries = self.chosen_entries() or ( + [self._single] if self._single else []) + return [e["spotify"] for e in entries if e.get("spotify")] + def chosen_entries(self) -> list[dict]: """The ticked songs, in the list's order. Empty means "download the link itself" (one song, or a link that was never listed).""" @@ -201,12 +217,15 @@ class UrlImportDialog(QDialog): self._single = None self._url.setEnabled(False) self._just.setVisible(bool(self._linked)) - self._status.setText("Checking what's at the link…") + spotify = self._kind == "spotify" + self._status.setText("Reading the songs off Spotify…" if spotify + else "Checking what's at the link…") self._status.show() self._busy.show() # No parent: a cancelled probe's thread can outlive this dialog, # and must not be emitting from a deleted QObject when it does. - probe = PlaylistProbe(url, cookies_browser=self._cookies) + probe = (SpotifyProbe if spotify else PlaylistProbe)( + url, cookies_browser=self._cookies) probe.entry.connect(self._on_entry) probe.finished.connect(self._on_probe_finished) probe.failed.connect(self._on_probe_failed) @@ -266,6 +285,11 @@ class UrlImportDialog(QDialog): def _count_text(self, listing: bool) -> str: n = len(self._entries) + if self._kind == "spotify": + return (f"This Spotify link has {n} songs" + + ("…" if listing else ".") + + "\nEach one is found on YouTube by its artist, title " + "and length.") text = f"This link is a playlist: {n} songs" if listing: text += " so far, still listing…" diff --git a/lintunes/spotify_link.py b/lintunes/spotify_link.py new file mode 100644 index 0000000..ac95f60 --- /dev/null +++ b/lintunes/spotify_link.py @@ -0,0 +1,222 @@ +"""Spotify links for File ▸ Import from URL… (Round 76). + +Spotify won't hand over audio, but it will say what a link *is*: its public +embed page (``open.spotify.com/embed//``) carries the whole track +list as JSON in a ``__NEXT_DATA__`` script, with no API key and no login. +Each song is then looked up on YouTube as ``Artist - Title`` and downloaded +with the `song` flags like any other link. Ported from trav's +``spotify-youtube`` script. + +Two things this does that the script didn't: + +- **Length picks the YouTube result.** Every embed track carries its + duration, so of the first five search results the one closest in length + wins. A length that disagrees is a live take, a cover, a lyric video with + a minute of intro, or another song with the same name. Words like "live" + and "cover" in a result's title count against it unless the Spotify title + says so too. +- **Spotify's tags go on before the import**, so the song files under its + real artist and album rather than ``Unknown Artist``, and Identify ranks + AcoustID's recordings by the artist Spotify named. + +Shaped like ``art_search.py``: pure helpers first (tested against canned +JSON), then the one network call. +""" +import json +import re +from dataclasses import dataclass + +import requests + +EMBED_URL = "https://open.spotify.com/embed/{kind}/{id}" +USER_AGENT = ("Mozilla/5.0 (X11; Linux x86_64; rv:131.0) " + "Gecko/20100101 Firefox/131.0") +TIMEOUT = 20 + +# How many YouTube results each song is chosen from. +SEARCH_RESULTS = 5 +# A result this much longer or shorter than Spotify's length is suspect. +DURATION_SLACK = 10 + +KINDS = ("track", "album", "playlist") + +# open.spotify.com/track/, …/intl-de/album/?si=…, spotify:playlist: +_SOURCE = re.compile( + r"^(?:https?://open\.spotify\.com/(?:intl-[\w-]+/)?(?:embed/)?" + r"(track|album|playlist)/([A-Za-z0-9]+)" + r"|spotify:(track|album|playlist):([A-Za-z0-9]+))(?:[/?#].*)?$") + +_NEXT_DATA = re.compile( + r'\s*(.*?)\s*' + r'', re.DOTALL) + +# Words in a YouTube title that mean "not the studio recording". +_UNWANTED = ("live", "cover", "karaoke", "instrumental", "remix", "reaction", + "8d", "slowed", "sped up", "nightcore") + + +class SpotifyError(Exception): + pass + + +@dataclass +class SpotifyTrack: + id: str + title: str + artist: str + album: str = "" + duration: int | None = None # seconds + track_number: int = 0 + track_count: int = 0 + + def label(self) -> str: + return f"{self.artist} — {self.title}" if self.artist else self.title + + +def spotify_source(text: str) -> tuple[str, str] | None: + """(kind, id) for a Spotify track, album or playlist link or URI, else + None. Artist, show and episode links aren't lists of songs.""" + m = _SOURCE.match((text or "").strip()) + if not m: + return None + return (m.group(1), m.group(2)) if m.group(1) else (m.group(3), + m.group(4)) + + +def _seconds(ms) -> int | None: + try: + return round(float(ms) / 1000) or None + except (TypeError, ValueError): + return None + + +def _track_id(item: dict, fallback: str) -> str: + uri = item.get("uri") or "" + return uri.rsplit(":", 1)[-1] if uri.startswith("spotify:track:") \ + else (item.get("id") or fallback) + + +def _find_entity(data, depth=0): + """The dict holding the track list (or the lone track), wherever the + page put it.""" + if depth > 20 or data is None: + return None + if isinstance(data, dict): + if isinstance(data.get("trackList"), list) and data["trackList"]: + return data + if data.get("type") == "track" and isinstance(data.get("artists"), + list): + return data + children = data.values() + elif isinstance(data, list): + children = data + else: + return None + for value in children: + found = _find_entity(value, depth + 1) + if found is not None: + return found + return None + + +def parse_embed(html: str) -> tuple[str, list[SpotifyTrack]]: + """An embed page → (its name, its songs in order). Raises SpotifyError + when the page isn't shaped the way it was in October 2026.""" + m = _NEXT_DATA.search(html or "") + if not m: + raise SpotifyError("Spotify's page didn't have the track list in it " + "(Spotify may have changed its page format)") + try: + data = json.loads(m.group(1)) + except ValueError as e: + raise SpotifyError(f"Spotify's track list couldn't be read: {e}") + entity = _find_entity(data) + if entity is None: + raise SpotifyError("Spotify's page had no songs in it") + + name = entity.get("name") or entity.get("title") or "Spotify" + if "trackList" not in entity: # a single track + artist = ", ".join(a.get("name", "") for a in entity["artists"] + if a.get("name")) + title = entity.get("title") or entity.get("name") or "" + if not title: + raise SpotifyError("Spotify's page had no songs in it") + return title, [SpotifyTrack( + id=_track_id(entity, entity.get("id") or "0"), title=title, + artist=artist, duration=_seconds(entity.get("duration")))] + + is_album = entity.get("type") == "album" + items = [item for item in entity["trackList"] if item.get("title")] + tracks = [] + for n, item in enumerate(items, 1): + tracks.append(SpotifyTrack( + id=_track_id(item, str(n)), title=item["title"], + artist=item.get("subtitle") or "", + album=name if is_album else "", + duration=_seconds(item.get("duration")), + track_number=n if is_album else 0, + track_count=len(items) if is_album else 0)) + if not tracks: + raise SpotifyError("Spotify's page had no songs in it") + return name, tracks + + +def search_query(track: SpotifyTrack) -> str: + return f"{track.artist} - {track.title}" if track.artist else track.title + + +def _unwanted(title: str, wanted_title: str) -> int: + """How many "not the studio take" words a result has that the song's + own title doesn't.""" + have = (title or "").lower() + want = (wanted_title or "").lower() + return sum(1 for word in _UNWANTED + if re.search(rf"\b{re.escape(word)}\b", have) + and not re.search(rf"\b{re.escape(word)}\b", want)) + + +def rank_candidates(entries: list[dict], + track: SpotifyTrack) -> list[dict]: + """YouTube search results (``PlaylistProbe`` entries), best first: within + ``DURATION_SLACK`` of Spotify's length before outside it, no "live" / + "cover" before those words, then closest in length, then YouTube's own + order. With no lengths on either side, YouTube's order stands.""" + def key(pair): + position, entry = pair + length = entry.get("duration") + if track.duration and length: + off = abs(length - track.duration) + return (off > DURATION_SLACK, + _unwanted(entry.get("title", ""), track.title), + off, position) + return (track.duration is not None, + _unwanted(entry.get("title", ""), track.title), 0, position) + return [e for _, e in sorted(enumerate(entries), key=key)] + + +def tag_fields(track: SpotifyTrack) -> dict: + """What Spotify knows, as Track-style fields for ``tagging.write_tags``.""" + fields = {"name": track.title, "artist": track.artist} + if track.album: + fields["album"] = track.album + if track.track_number: + fields["track_number"] = track.track_number + fields["track_count"] = track.track_count + return fields + + +# ---- network ---- + +def fetch(kind: str, source_id: str) -> tuple[str, list[SpotifyTrack]]: + url = EMBED_URL.format(kind=kind, id=source_id) + try: + resp = requests.get(url, headers={"User-Agent": USER_AGENT}, + timeout=TIMEOUT) + except requests.RequestException as e: + raise SpotifyError(f"couldn't reach Spotify: {e}") + if resp.status_code == 404: + raise SpotifyError(f"Spotify has no {kind} with that link (it may be " + "private or deleted)") + if not resp.ok: + raise SpotifyError(f"Spotify answered {resp.status_code}") + return parse_embed(resp.text) diff --git a/lintunes/url_import.py b/lintunes/url_import.py index 90e7135..e7c2504 100644 --- a/lintunes/url_import.py +++ b/lintunes/url_import.py @@ -31,6 +31,13 @@ Anything but a plain song link is *listed* before it downloads ``list=RD…``, and yt-dlp's default is to take the whole list. The songs the user picks are then downloaded by their own pages, never by playlist position, because a Mix comes back reshuffled every time it's listed. + +A Spotify link (Round 76) can't be downloaded at all, but its embed page +names every song on it (``spotify_link``). ``SpotifyProbe`` lists those for +the same checklist, and ``SpotifyImportWorker`` takes them one at a time: +a five-result YouTube search, the result closest to Spotify's length, the +`song` download, then Spotify's tags written into the file *before* it's +handed over, so the import files it under the right artist and album. """ import os import shutil @@ -38,6 +45,7 @@ import signal import subprocess import re import tempfile +import logging import threading from dataclasses import dataclass from pathlib import Path @@ -45,8 +53,11 @@ from urllib.parse import parse_qs, urlsplit from PyQt6.QtCore import QObject, pyqtSignal +from lintunes import spotify_link, tagging from lintunes.models.playlist import PlaylistType +log = logging.getLogger(__name__) + # The `song` helper's own flags, verbatim. SONG_ARGS = ["--extract-audio", "--audio-format", "mp3"] @@ -90,6 +101,8 @@ def ytdlp_available() -> bool: def looks_like_url(text: str) -> bool: text = (text or "").strip() + if spotify_link.spotify_source(text) is not None: + return True # spotify:track:… URIs too return (text.startswith(("http://", "https://")) and len(text) > len("https://") and not any(c.isspace() for c in text)) @@ -98,12 +111,16 @@ def looks_like_url(text: str) -> bool: def link_kind(url: str) -> str: """What a link names, from the URL alone: "song", "song_in_list" (a song opened from a playlist or a YouTube Mix — yt-dlp would take the whole - list), "list", or "unknown" (another site; only a probe can tell). + list), "list", "spotify" (a Spotify track, album or playlist, which is + found on YouTube song by song), or "unknown" (another site; only a probe + can tell). Only "song" skips the probe, so a plain song link imports as fast as ever. A Mix is the one that bites: it's a song link with ``list=RD…`` on the end, and the list behind it runs to hundreds of songs. """ + if spotify_link.spotify_source(url) is not None: + return "spotify" try: parts = urlsplit((url or "").strip()) except ValueError: @@ -364,6 +381,7 @@ class _YtdlpRun(QObject): return None count, errors, returncode = attempt if count: + self._cookies = cookies # a later run starts with them self.cookies_used.emit(cookies) return count, errors, cookies, returncode @@ -539,3 +557,169 @@ class UrlImportWorker(_YtdlpRun): return self.finished.emit({"downloaded": count, "errors": errors, "cancelled": cancelled}) + + +# ---- Spotify ---- + +class SpotifyProbe(QObject): + """``PlaylistProbe``'s signals for a Spotify link: reads its embed page + on a daemon thread and emits one entry per song, carrying the + ``SpotifyTrack`` under ``"spotify"``. One request, no yt-dlp.""" + + entry = pyqtSignal(dict) + finished = pyqtSignal(int) + failed = pyqtSignal(str) + cookies_used = pyqtSignal(str) # never: Spotify needs no cookies + + def __init__(self, url: str, parent=None, **_kw): + super().__init__(parent) + self._url = url + self._busy = False + self._cancel = threading.Event() + + def busy(self) -> bool: + return self._busy + + def cancel(self): + self._cancel.set() + + def start(self): + if not self._busy: + self._busy = True + threading.Thread(target=self._run, daemon=True).start() + + def _run(self): + try: + kind, source_id = spotify_link.spotify_source(self._url) + _name, tracks = spotify_link.fetch(kind, source_id) + except Exception as e: + self._busy = False + self.failed.emit(str(e)) + return + count = 0 + for track in tracks: + if self._cancel.is_set(): + break + self.entry.emit({"id": track.id, "title": track.label(), + "duration": track.duration, "url": "", + "spotify": track}) + count += 1 + self._busy = False + self.finished.emit(count) + + +class SpotifyImportWorker(UrlImportWorker): + """``UrlImportWorker``'s signals, for songs a Spotify link named: each + is searched for on YouTube, the best-length result downloaded (the next + one if that fails), and Spotify's tags written in before ``downloaded``. + Rows are keyed by the Spotify id, which is what the checklist and the + progress window know them by.""" + + # Results tried per song before giving up on it. + TRIES = 3 + + def __init__(self, tracks, parent=None, *, + cookies_browser: str | None = None): + super().__init__([], parent, cookies_browser=cookies_browser) + self._tracks = list(tracks) + self._stage = "" + self._query = "" + self._found: list[dict] = [] + self._files: list[str] = [] + + def _command(self, cookies_browser): + if self._stage == "search": + return build_probe_command( + f"ytsearch{spotify_link.SEARCH_RESULTS}:{self._query}", + cookies_browser) + return build_command(self._query, self.temp_dir, cookies_browser) + + def _on_parsed(self, parsed): + kind = parsed[0] + if self._stage == "search": + if kind == "entry": + self._found.append(parsed[1]) + return True + elif kind == "progress": + self.progress.emit(parsed[1]) + elif kind == "file" and not self._cancel.is_set(): + self._files.append(parsed[1]) + return True + return False + + def _on_error(self, error): + pass # reported per song, by Spotify id + + def _search(self, track): + """→ (YouTube results best first, ERROR lines, cookies, code), or + None when yt-dlp couldn't run.""" + self._stage, self._query, self._found = ("search", + spotify_link.search_query( + track), []) + result = self._run_with_retry() + if result is None: + return None + _count, errors, cookies, code = result + return (spotify_link.rank_candidates(self._found, track), errors, + cookies, code) + + def _download(self, url): + self._stage, self._query, self._files = "download", url, [] + result = self._run_with_retry() + if result is None: + return None + return self._files[-1] if self._files else "", result[1] + + def _run(self): + total = len(self._tracks) + downloaded = 0 + errors: list[str] = [] + last = None # (ERROR lines, cookies, code) of a miss + for index, track in enumerate(self._tracks, 1): + if self._cancel.is_set(): + break + self.item_started.emit(index, total, track.id, track.label()) + searched = self._search(track) + if searched is None: + return + candidates, run_errors, cookies, code = searched + path, reason = "", "" + if not candidates: + reason = (run_errors[-1] if run_errors + else "no match on YouTube") + last = (run_errors, cookies, code) + for candidate in candidates[:self.TRIES]: + if self._cancel.is_set(): + break + got = self._download(candidate["url"]) + if got is None: + return + path, run_errors = got + if path: + break + reason = run_errors[-1] if run_errors else "download failed" + last = (run_errors, self._cookies, 0) + if self._cancel.is_set() and not path: + break + if not path: + errors.append(f"{track.label()}: {reason}") + self.item_failed.emit(track.id, reason) + continue + try: + tagging.write_tags(path, spotify_link.tag_fields(track)) + except Exception as e: # untagged still beats not imported + log.warning("spotify tags for %s: %s", path, e) + downloaded += 1 + self.downloaded.emit(path) + cancelled = self._cancel.is_set() + self._busy = False + if downloaded == 0 and not cancelled: + if last is not None and last[0]: + self._fail_nothing(*last, "download") + else: + self.failed.emit(errors[-1] if errors else + "nothing on the Spotify link was found " + "on YouTube") + return + self.finished.emit({"downloaded": downloaded, "errors": errors, + "cancelled": cancelled}) diff --git a/tests/test_round49.py b/tests/test_round49.py index 0dd4de2..b4fcef4 100644 --- a/tests/test_round49.py +++ b/tests/test_round49.py @@ -337,6 +337,8 @@ class TestWindow: assert "T2" in description def chosen_entries(self): return [] + def spotify_tracks(self): + return [] def download_urls(self): return [self.url()] def exec(self): diff --git a/tests/test_round75.py b/tests/test_round75.py index 0eec876..1492bd3 100644 --- a/tests/test_round75.py +++ b/tests/test_round75.py @@ -362,6 +362,8 @@ def test_only_the_chosen_songs_download(window, qapp, fake, mp3_file, return False def chosen_entries(self): return chosen + def spotify_tracks(self): + return [] def download_urls(self): return [c["url"] for c in chosen] monkeypatch.setattr(mw, "UrlImportDialog", _Dialog) @@ -406,6 +408,8 @@ def test_one_song_gets_no_progress_window(window, qapp, fake, mp3_file, return False def chosen_entries(self): return [] + def spotify_tracks(self): + return [] def download_urls(self): return [self.url()] monkeypatch.setattr(mw, "UrlImportDialog", _Dialog) diff --git a/tests/test_round76.py b/tests/test_round76.py new file mode 100644 index 0000000..d639fc3 --- /dev/null +++ b/tests/test_round76.py @@ -0,0 +1,363 @@ +"""Round 76: Spotify links in File ▸ Import from URL. + +trav's ``spotify-youtube`` script, brought inside: a Spotify track, album or +playlist link is read off Spotify's public embed page, each song is found on +YouTube (the result closest to Spotify's length wins), downloaded with the +`song` flags, tagged with what Spotify said, then imported and identified +like any other link. No network here: the page is canned and yt-dlp is fake. +""" +import json +import os +import stat +import sys +import time + +import pytest + +from lintunes import spotify_link, tagging, url_import +from lintunes.gui.url_import_dialog import UrlImportDialog +from lintunes.library_manager import LibraryManager +from lintunes.models import Library +from lintunes.preferences import Preferences +from lintunes.spotify_link import ( + SpotifyError, SpotifyTrack, parse_embed, rank_candidates, spotify_source, +) + +ALBUM_URL = "https://open.spotify.com/album/57F44c0MTziVzHPEuJtH9A?si=xyz" + + +def _page(entity) -> str: + data = {"props": {"pageProps": {"state": {"data": {"entity": entity}}}}} + return ('") + + +ALBUM = _page({ + "type": "album", "name": "Last Splash", "title": "Last Splash", + "subtitle": "The Breeders", + "trackList": [ + {"uri": "spotify:track:t1", "title": "New Year", + "subtitle": "The Breeders", "duration": 116706}, + {"uri": "spotify:track:t2", "title": "Cannonball", + "subtitle": "The Breeders", "duration": 213000}, + {"uri": "spotify:track:t3", "title": "", # dropped + "subtitle": "The Breeders", "duration": 1000}, + ]}) + +PLAYLIST = _page({ + "type": "playlist", "name": "Chill", + "trackList": [ + {"uri": "spotify:track:p1", "title": "Song A", "subtitle": "X, Y", + "duration": 200000}]}) + +TRACK = _page({ + "type": "track", "name": "Never Gonna Give You Up", + "title": "Never Gonna Give You Up", "id": "4uLU", + "uri": "spotify:track:4uLU", + "artists": [{"name": "Rick Astley"}], "duration": 213573}) + + +# -------------------------------------------------------------------------- +# A. reading a link + + +@pytest.mark.parametrize("text, source", [ + (ALBUM_URL, ("album", "57F44c0MTziVzHPEuJtH9A")), + ("https://open.spotify.com/intl-de/track/4uLU", ("track", "4uLU")), + ("https://open.spotify.com/embed/playlist/37i9", ("playlist", "37i9")), + ("spotify:playlist:37i9", ("playlist", "37i9")), + ("https://open.spotify.com/artist/0gxy", None), + ("https://open.spotify.com/episode/0gxy", None), + ("https://www.youtube.com/watch?v=abc", None), + ("", None), +]) +def test_spotify_source(text, source): + assert spotify_source(text) == source + + +def test_link_kind_and_uris_count_as_links(): + assert url_import.link_kind(ALBUM_URL) == "spotify" + assert url_import.link_kind("spotify:track:4uLU") == "spotify" + assert url_import.looks_like_url("spotify:track:4uLU") + assert url_import.link_kind("https://open.spotify.com/artist/x") \ + == "unknown" + + +def test_an_album_page_numbers_its_tracks(): + name, tracks = parse_embed(ALBUM) + assert name == "Last Splash" + assert [(t.id, t.title, t.album, t.track_number, t.track_count, + t.duration) for t in tracks] == [ + ("t1", "New Year", "Last Splash", 1, 2, 117), + ("t2", "Cannonball", "Last Splash", 2, 2, 213)] + + +def test_a_playlist_page_has_no_album(): + _, [track] = parse_embed(PLAYLIST) + assert (track.artist, track.album, track.track_number) == ("X, Y", "", 0) + assert spotify_link.tag_fields(track) == {"name": "Song A", + "artist": "X, Y"} + + +def test_a_track_page_is_one_song(): + name, [track] = parse_embed(TRACK) + assert name == "Never Gonna Give You Up" + assert (track.id, track.artist, track.duration) == ("4uLU", + "Rick Astley", 214) + + +@pytest.mark.parametrize("html", [ + "no data", + _page({"type": "playlist", "name": "Empty", "trackList": []}), +]) +def test_a_changed_page_says_so(html): + with pytest.raises(SpotifyError): + parse_embed(html) + + +# -------------------------------------------------------------------------- +# B. choosing the YouTube result + + +def _e(vid, duration, title="x"): + return {"id": vid, "duration": duration, "title": title, + "url": f"https://www.youtube.com/watch?v={vid}"} + + +def test_the_closest_length_wins(): + track = SpotifyTrack("t", "Cannonball", "The Breeders", duration=213) + ranked = rank_candidates([_e("video", 260), _e("audio", 214), + _e("lyric", 220)], track) + assert [e["id"] for e in ranked] == ["audio", "lyric", "video"] + + +def test_live_and_covers_lose_unless_the_song_is_one(): + track = SpotifyTrack("t", "Cannonball", "The Breeders", duration=213) + ranked = rank_candidates([_e("live", 213, "Cannonball (Live 1994)"), + _e("studio", 216, "Cannonball")], track) + assert ranked[0]["id"] == "studio" + live = SpotifyTrack("t", "Cannonball - Live", "The Breeders", + duration=213) + assert rank_candidates([_e("live", 213, "Cannonball (Live 1994)"), + _e("studio", 216, "Cannonball")], + live)[0]["id"] == "live" + + +def test_without_lengths_youtube_order_stands(): + track = SpotifyTrack("t", "Song", "A") + assert [e["id"] for e in rank_candidates( + [_e("a", None), _e("b", 100)], track)] == ["a", "b"] + + +# -------------------------------------------------------------------------- +# C. the worker, against a fake yt-dlp + +# Search: "ytsearch5:" lists $FAKE_SEARCH[query] as (id, seconds, +# title). Download: writes ".mp3", failing ids starting with "bad". +FAKE_YTDLP = '''#!/usr/bin/env python3 +import json, os, sys +args = sys.argv[1:] +with open(os.environ["FAKE_YTDLP_LOG"], "a") as log: + log.write("\\x1f".join(args) + "\\n") +urls = args[args.index("--") + 1:] +if "--flat-playlist" in args: + query = urls[0].split(":", 1)[1] + for vid, secs, title in json.loads(os.environ["FAKE_SEARCH"]).get(query, []): + print(f"LTENTRY {vid}\\t{secs}\\t" + f"https://www.youtube.com/watch?v={vid}\\t{title}", flush=True) + sys.exit(0) +dest = args[args.index("-P") + 1] +for url in urls: + vid = url.rsplit("=", 1)[-1] + if vid.startswith("bad"): + print(f"ERROR: [youtube] {vid}: Video unavailable", file=sys.stderr, + flush=True) + continue + print(f"LTSTART 1\\t1\\t{vid}\\t{vid}", flush=True) + print("LTPROG 50.0%", flush=True) + path = os.path.join(dest, f"{vid} [{vid}].mp3") + with open(path, "wb") as f: + f.write(open(os.environ["FAKE_YTDLP_AUDIO"], "rb").read()) + print(f"LTFILE {path}", flush=True) +''' + + +@pytest.fixture +def fake(tmp_path, monkeypatch, mp3_file): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + script = bin_dir / "yt-dlp" + script.write_text(FAKE_YTDLP.replace("/usr/bin/env python3", + sys.executable)) + script.chmod(script.stat().st_mode | stat.S_IEXEC) + log = tmp_path / "argv.log" + monkeypatch.setenv("PATH", f"{bin_dir}{os.pathsep}{os.environ['PATH']}") + monkeypatch.setenv("FAKE_YTDLP_LOG", str(log)) + monkeypatch.setenv("FAKE_YTDLP_AUDIO", str(mp3_file)) + monkeypatch.setenv("XDG_CONFIG_HOME", str(tmp_path / "config")) + + def search(results): + monkeypatch.setenv("FAKE_SEARCH", json.dumps(results)) + search({}) + search.argv = lambda: ([line.split("\x1f") for line in + log.read_text().splitlines()] + if log.exists() else []) + return search + + +def _worker(tmp_path, tracks): + worker = url_import.SpotifyImportWorker(tracks) + worker.temp_dir = tmp_path / "dl" + worker.temp_dir.mkdir() + events = [] + worker.item_started.connect( + lambda i, n, sid, t: events.append(("start", i, n, sid))) + worker.item_failed.connect(lambda sid, why: events.append(("bad", sid))) + worker.downloaded.connect(lambda p: events.append(("file", p))) + worker.finished.connect(lambda d: events.append(("done", d["downloaded"]))) + worker.failed.connect(lambda m: events.append(("failed", m))) + return worker, events + + +def test_each_song_is_searched_picked_downloaded_and_tagged(fake, tmp_path): + _, tracks = parse_embed(ALBUM) + fake({"The Breeders - New Year": [["vid1", 300, "New Year (video)"], + ["aud1", 117, "New Year"]], + "The Breeders - Cannonball": [["bad2", 213, "Cannonball"], + ["aud2", 214, "Cannonball"]]}) + worker, events = _worker(tmp_path, tracks) + worker._run() + + files = [e[1] for e in events if e[0] == "file"] + assert [e for e in events if e[0] != "file"] == [ + ("start", 1, 2, "t1"), ("start", 2, 2, "t2"), ("done", 2)] + # The right length, then the next result when the best one fails. + assert [os.path.basename(f) for f in files] == ["aud1 [aud1].mp3", + "aud2 [aud2].mp3"] + tags = tagging.read_tags(files[1]) + assert (tags["name"], tags["artist"], tags["album"], + tags["track_number"], tags["track_count"]) == ( + "Cannonball", "The Breeders", "Last Splash", 2, 2) + downloads = [a for a in fake.argv() if "--flat-playlist" not in a] + assert all("--extract-audio" in a for a in downloads) # the `song` flags + + +def test_a_song_youtube_doesnt_have_is_named(fake, tmp_path): + _, tracks = parse_embed(ALBUM) + fake({"The Breeders - Cannonball": [["aud2", 214, "Cannonball"]]}) + worker, events = _worker(tmp_path, tracks) + worker._run() + assert [e for e in events if e[0] != "file"] == [ + ("start", 1, 2, "t1"), ("bad", "t1"), ("start", 2, 2, "t2"), + ("done", 1)] + + +def test_nothing_found_at_all_fails(fake, tmp_path): + _, tracks = parse_embed(PLAYLIST) + worker, events = _worker(tmp_path, tracks) + worker._run() + assert events[-1][0] == "failed" + assert not worker.busy() + + +# -------------------------------------------------------------------------- +# D. the dialog and the window + + +def _pump(qapp, until, timeout=10.0): + deadline = time.monotonic() + timeout + while not until() and time.monotonic() < deadline: + qapp.processEvents() + time.sleep(0.01) + assert until(), "timed out waiting" + + +@pytest.fixture +def canned(monkeypatch): + pages = {"album": ALBUM, "track": TRACK, "playlist": PLAYLIST} + monkeypatch.setattr(spotify_link, "fetch", + lambda kind, _id: parse_embed(pages[kind])) + + +def _dialog(qapp, url): + qapp.clipboard().setText(url) + dialog = UrlImportDialog(None) + assert dialog.url() == url + return dialog + + +def test_an_album_is_a_checklist_of_spotify_songs(qapp, canned): + dialog = _dialog(qapp, ALBUM_URL) + dialog._on_import() + _pump(qapp, lambda: dialog._probe is None) + assert dialog._picking and dialog._list.count() == 2 + assert dialog._ok.text() == "Import 2 Songs" + dialog._list.item(0).setCheckState( + dialog._list.item(0).checkState().Unchecked) + dialog._on_import() + assert [t.title for t in dialog.spotify_tracks()] == ["Cannonball"] + + +def test_a_spotify_track_goes_straight_through(qapp, canned): + dialog = _dialog(qapp, "https://open.spotify.com/track/4uLU") + dialog._on_import() + _pump(qapp, lambda: dialog.result() == dialog.DialogCode.Accepted) + assert [t.artist for t in dialog.spotify_tracks()] == ["Rick Astley"] + + +def test_a_youtube_link_has_no_spotify_tracks(qapp): + dialog = _dialog(qapp, "https://www.youtube.com/watch?v=abc") + dialog._on_import() + assert dialog.spotify_tracks() == [] + + +@pytest.fixture +def window(qapp, tmp_path, fake, monkeypatch): + from lintunes.gui.main_window import MainWindow + library = Library(music_folder=str(tmp_path / "media")) + (tmp_path / "media").mkdir() + manager = LibraryManager(library, tmp_path / "data") + win = MainWindow(manager, Preferences(tmp_path / "data")) + identified = [] + monkeypatch.setattr(win, "_enqueue_identify", identified.extend) + win.identified = identified + yield win + win.close() + + +def test_spotify_songs_import_tagged_and_go_to_identify(window, qapp, fake, + monkeypatch): + from lintunes.gui import main_window as mw + _, tracks = parse_embed(ALBUM) + fake({"The Breeders - New Year": [["aud1", 117, "New Year"]], + "The Breeders - Cannonball": [["aud2", 214, "Cannonball"]]}) + chosen = [{"id": t.id, "title": t.label(), "spotify": t} + for t in tracks] + + class _Dialog: + cookies_used = None + def __init__(self, *a, **kw): + pass + def exec(self): + return True + def add_to_playlist(self): + return False + def chosen_entries(self): + return chosen + def spotify_tracks(self): + return tracks + def download_urls(self): + raise AssertionError("a Spotify link is never handed to yt-dlp") + monkeypatch.setattr(mw, "UrlImportDialog", _Dialog) + + window._import_from_url() + worker = window._url_worker + assert isinstance(worker, url_import.SpotifyImportWorker) + _pump(qapp, lambda: worker.temp_dir is None) + assert window._url_progress.status("t1") == "Imported" + assert window._url_progress.status("t2") == "Imported" + lib = window._manager.library.tracks.values() + assert sorted((t.artist, t.album, t.name) for t in lib) == [ + ("The Breeders", "Last Splash", "Cannonball"), + ("The Breeders", "Last Splash", "New Year")] + assert len(window.identified) == 2