Files
lintunes/lintunes/url_import.py
T
travandClaude Opus 5.5 581be8d732 v0.37.0: Import from URL takes Spotify links
A Spotify track, album or playlist is read off Spotify's public embed
page, each song found on YouTube by the result closest to Spotify's
length, downloaded with the song flags, tagged with Spotify's
artist/title/album/track number, then imported and identified.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-09 01:09:46 -07:00

726 lines
27 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""File ▸ Import from URL… — trav's `song` shell helper, run from inside LinTunes.
`song` is ``yt-dlp --extract-audio --audio-format mp3 "$@"``. This module runs
exactly those flags; everything else on the command line is plumbing for
reading back what yt-dlp did. yt-dlp is a *runtime* CLI tool, detected with
``shutil.which`` like fpcalc and ffmpeg, never a pip dependency.
Shaped like ``fingerprint.py``: pure helpers first (the command line, the
marker parser, and where the songs should go), then the subprocess on a daemon
thread behind ``UrlImportWorker``.
Downloads land in a private temp dir and are handed to the GUI one at a time,
which imports each as it arrives. The normal importer *copies* it into the
organized tree, so the temp dir only ever holds our own downloads. The GUI
deletes it (``cleanup``) once it has seen ``finished`` or ``failed``: those
signals are queued after every ``downloaded``, so by then every file has been
copied. yt-dlp's default filename, ``Title [id].mp3``, is kept on purpose,
because it's exactly what ``filename_tags.parse_filename`` reads.
YouTube sometimes answers an anonymous request with "Sign in to confirm
you're not a bot" (it flags the *network*, so cell connections see it
most). The cure yt-dlp names is the browser's own YouTube cookies. So a run
that downloads nothing because of that check is retried once with
``--cookies-from-browser``, and the GUI remembers the browser in this
machine's ``config.json`` (``ytdlp_cookies_browser``) so later imports send
the cookies from the start. Which browser holds a YouTube login is a fact
about this machine, which is why it isn't in the synced preferences.
Anything but a plain song link is *listed* before it downloads
(``link_kind``, ``PlaylistProbe``): a song opened from a YouTube Mix carries
``list=RD…``, and yt-dlp's default is to take the whole list. The songs the
user picks are then downloaded by their own pages, never by playlist
position, because a Mix comes back reshuffled every time it's listed.
A Spotify link (Round 76) can't be downloaded at all, but its embed page
names every song on it (``spotify_link``). ``SpotifyProbe`` lists those for
the same checklist, and ``SpotifyImportWorker`` takes them one at a time:
a five-result YouTube search, the result closest to Spotify's length, the
`song` download, then Spotify's tags written into the file *before* it's
handed over, so the import files it under the right artist and album.
"""
import os
import shutil
import signal
import subprocess
import re
import tempfile
import logging
import threading
from dataclasses import dataclass
from pathlib import Path
from urllib.parse import parse_qs, urlsplit
from PyQt6.QtCore import QObject, pyqtSignal
from lintunes import spotify_link, tagging
from lintunes.models.playlist import PlaylistType
log = logging.getLogger(__name__)
# The `song` helper's own flags, verbatim.
SONG_ARGS = ["--extract-audio", "--audio-format", "mp3"]
# Line prefixes we ask yt-dlp to print, so its output can be read reliably
# instead of scraped.
START = "LTSTART"
FILE = "LTFILE"
PROGRESS = "LTPROG"
ENTRY = "LTENTRY"
# What YouTube says when it wants a signed-in browser. Matched on a fragment
# that sidesteps the apostrophe, which yt-dlp prints as a curly ’.
BOT_CHECK = "to confirm you"
# The config.json key naming the browser yt-dlp borrows cookies from.
COOKIES_KEY = "ytdlp_cookies_browser"
def is_bot_check(error: str) -> bool:
return BOT_CHECK in (error or "") and "bot" in error
def default_cookies_browser() -> str | None:
"""The browser whose cookies to borrow: the first one with a profile on
this machine, or None when there's nothing to borrow from."""
home = Path.home()
config = Path(os.environ.get("XDG_CONFIG_HOME", str(home / ".config")))
for browser, profile in (("firefox", home / ".mozilla" / "firefox"),
("chromium", config / "chromium"),
("chrome", config / "google-chrome")):
if profile.is_dir():
return browser
return None
def ytdlp_available() -> bool:
"""Whether the yt-dlp *binary* is on PATH (the fpcalc_available pattern)."""
return shutil.which("yt-dlp") is not None
def looks_like_url(text: str) -> bool:
text = (text or "").strip()
if spotify_link.spotify_source(text) is not None:
return True # spotify:track:… URIs too
return (text.startswith(("http://", "https://"))
and len(text) > len("https://")
and not any(c.isspace() for c in text))
def link_kind(url: str) -> str:
"""What a link names, from the URL alone: "song", "song_in_list" (a song
opened from a playlist or a YouTube Mix — yt-dlp would take the whole
list), "list", "spotify" (a Spotify track, album or playlist, which is
found on YouTube song by song), or "unknown" (another site; only a probe
can tell).
Only "song" skips the probe, so a plain song link imports as fast as
ever. A Mix is the one that bites: it's a song link with ``list=RD…``
on the end, and the list behind it runs to hundreds of songs.
"""
if spotify_link.spotify_source(url) is not None:
return "spotify"
try:
parts = urlsplit((url or "").strip())
except ValueError:
return "unknown"
host = (parts.hostname or "").lower()
if host.startswith("www."):
host = host[4:]
if host.startswith("m."):
host = host[2:]
query = parse_qs(parts.query)
has_list = bool(query.get("list"))
if host == "youtu.be":
return "song_in_list" if has_list else "song"
if host in ("youtube.com", "music.youtube.com"):
if parts.path == "/watch" and query.get("v"):
return "song_in_list" if has_list else "song"
if parts.path.startswith("/shorts/"):
return "song"
if parts.path == "/playlist" and has_list:
return "list"
return "unknown"
def build_command(urls, dest_dir,
cookies_browser: str | None = None) -> list[str]:
"""yt-dlp with the `song` flags, downloading ``urls`` (one link, or a
list of them) into ``dest_dir``.
``before_dl`` names each item as it starts (its place in a playlist, its
id, its title),
``after_move`` gives the final mp3's path once ffmpeg has finished with it,
and the progress template reduces the progress bar to one number per line.
yt-dlp's default already carries on past an unavailable video in a
playlist, so "import everything at the link" needs no extra flag.
"""
if isinstance(urls, str):
urls = [urls]
return [
"yt-dlp", *SONG_ARGS,
*_cookie_args(cookies_browser),
"-P", str(dest_dir),
"--print",
f"before_dl:{START} %(playlist_index|1)s\t%(n_entries|1)s\t"
f"%(id)s\t%(title)s",
"--print", f"after_move:{FILE} %(filepath)s",
"--progress", "--newline",
"--progress-template", f"download:{PROGRESS} %(progress._percent_str)s",
"--", *urls,
]
def build_probe_command(url: str,
cookies_browser: str | None = None) -> list[str]:
"""List what's at a link without downloading any of it: one ENTRY line
per song, printed as yt-dlp finds it. ``webpage_url`` is each song's own
page, for a playlist entry and a lone song alike, and the download is
then handed those pages, never playlist positions: a Mix comes back in
a different order every time it's listed."""
return [
"yt-dlp", "--flat-playlist", *_cookie_args(cookies_browser),
"--print",
f"{ENTRY} %(id)s\t%(duration|)s\t%(webpage_url|)s\t%(title)s",
"--", url,
]
def _cookie_args(browser: str | None) -> list[str]:
return ["--cookies-from-browser", browser] if browser else []
def _int(text: str, default: int) -> int:
try:
return int(text)
except (TypeError, ValueError):
return default
def _duration(text: str) -> int | None:
try:
return int(float(text))
except (TypeError, ValueError):
return None
def parse_line(line: str):
"""One line of yt-dlp output → ("start", i, n, id, title) |
("file", path) | ("progress", percent) | ("entry", {id, duration, url,
title}) | None for everything else."""
line = line.rstrip("\r\n")
if line.startswith(START + " "):
parts = line[len(START) + 1:].split("\t", 3)
if len(parts) != 4:
return None
return ("start", _int(parts[0], 1), _int(parts[1], 1), parts[2],
parts[3])
if line.startswith(ENTRY + " "):
parts = line[len(ENTRY) + 1:].split("\t", 3)
if len(parts) != 4 or not parts[2].strip():
return None
return ("entry", {"id": parts[0], "duration": _duration(parts[1]),
"url": parts[2], "title": parts[3]})
if line.startswith(FILE + " "):
path = line[len(FILE) + 1:]
return ("file", path) if path.strip() else None
if line.startswith(PROGRESS + " "):
text = line[len(PROGRESS) + 1:].strip().rstrip("%")
try:
return ("progress", max(0, min(100, int(float(text)))))
except ValueError:
return None
return None
# "ERROR: [youtube] <id>: Video unavailable" — the id says which song.
_ERROR_ID = re.compile(r"^\[[^\]]+\] ([\w-]+): (.+)$")
def error_item(error: str) -> tuple[str, str] | None:
"""(id, reason) for an error yt-dlp pinned on one song, else None."""
m = _ERROR_ID.match(error or "")
return (m.group(1), m.group(2)) if m else None
# ---- where the songs go ----
@dataclass
class ImportTarget:
pid: str
position: int | None # playlist index to insert before; None appends
description: str # what the dialog tells the user
def accepts_adds(playlist) -> bool:
"""A playlist you can add songs to by hand: not a folder, not smart (its
membership comes from its rules), not a system list."""
return (playlist is not None
and playlist.playlist_type == PlaylistType.REGULAR
and not playlist.is_system)
def resolve_target(library, shown_pid: str, selected_row: int | None,
playing_context: str) -> ImportTarget | None:
"""Where "and add to current playlist?" puts the songs.
1. A playlist is shown and a song in it is selected: above that song.
2. Otherwise, the playlist that's playing: at its end.
3. Otherwise, the playlist that's shown: at its end.
4. Otherwise nowhere, and the checkbox is grayed out.
``selected_row`` is the selected song's row in the table's *source* model,
which is playlist order whatever the column sort is. That model skips
ids with no track behind them, so the row is mapped back to a real index
into ``track_ids``.
"""
shown = library.playlists.get(shown_pid) if shown_pid else None
if not accepts_adds(shown):
shown = None
if shown is not None and selected_row is not None:
present = [i for i, tid in enumerate(shown.track_ids)
if tid in library.tracks]
if 0 <= selected_row < len(present):
index = present[selected_row]
song = library.tracks[shown.track_ids[index]].name or "(untitled)"
return ImportTarget(shown.persistent_id, index,
f"Adds above “{song}” in “{shown.name}”")
if playing_context.startswith("playlist:"):
playing = library.playlists.get(playing_context.split(":", 1)[1])
if accepts_adds(playing):
return ImportTarget(playing.persistent_id, None,
f"Adds to the end of “{playing.name}” "
f"(playing)")
if shown is not None:
return ImportTarget(shown.persistent_id, None,
f"Adds to the end of “{shown.name}”")
return None
# ---- running yt-dlp ----
class _YtdlpRun(QObject):
"""What the probe and the download share: one yt-dlp at a time on a
daemon thread (ExportWorker's threading, not a QThread), a cancel that
takes ffmpeg down with it, and the bot-check retry with a browser's
cookies. Subclasses supply the command and read the marker lines."""
failed = pyqtSignal(str)
cookies_used = pyqtSignal(str)
def __init__(self, parent=None, *, cookies_browser: str | None = None):
super().__init__(parent)
self._cookies = cookies_browser
self._busy = False
self._cancel = threading.Event()
self._proc = None
def busy(self) -> bool:
return self._busy
def cancel(self):
self._cancel.set()
self._terminate()
def cancelled(self) -> bool:
return self._cancel.is_set()
def _terminate(self):
proc = self._proc
if proc is None or proc.poll() is not None:
return
# yt-dlp runs ffmpeg as a child. Signal the whole session, so a cancel
# during conversion doesn't leave ffmpeg writing into the temp dir.
try:
os.killpg(proc.pid, signal.SIGTERM)
except (OSError, AttributeError):
proc.terminate()
def _start_thread(self):
self._busy = True
threading.Thread(target=self._run_guarded, daemon=True).start()
def _run_guarded(self):
try:
self._run()
except Exception as e: # never leave the GUI waiting forever
self._busy = False
self.failed.emit(str(e))
# -- subclass hooks --
def _command(self, cookies_browser) -> list[str]:
raise NotImplementedError
def _on_parsed(self, parsed) -> bool:
"""One marker line. True when it counts as a result (a song found,
a song downloaded), which is what decides a bot-check retry."""
raise NotImplementedError
def _on_error(self, error: str):
pass
def _run_with_retry(self):
"""→ (results, ERROR lines, cookies used, exit code), or None when
yt-dlp couldn't be started (``failed`` has been emitted)."""
cookies = self._cookies
attempt = self._attempt(cookies)
if attempt is None:
return None
count, errors, returncode = attempt
if (count == 0 and not self._cancel.is_set() and not cookies
and any(is_bot_check(e) for e in errors)):
# Nothing landed, so a second pass can't duplicate a song.
cookies = default_cookies_browser()
if cookies:
attempt = self._attempt(cookies)
if attempt is None:
return None
count, errors, returncode = attempt
if count:
self._cookies = cookies # a later run starts with them
self.cookies_used.emit(cookies)
return count, errors, cookies, returncode
def _fail_nothing(self, errors, cookies, returncode, what: str):
"""Nothing came back: say why, in words a bot check deserves."""
if errors and is_bot_check(errors[-1]):
self.failed.emit(
"YouTube wants to see a signed-in browser before it "
"will hand this over"
+ (f" (tried {cookies}'s cookies)" if cookies else "")
+ ".\n\nSign in to YouTube in "
+ (cookies.title() if cookies else "your browser")
+ " and try again.\n\nyt-dlp said: " + errors[-1])
return
self.failed.emit(
errors[-1] if errors else
f"yt-dlp found nothing to {what} (exit code {returncode})")
def _attempt(self, cookies_browser):
"""One yt-dlp run → (results, ERROR lines, exit code), or None when
yt-dlp couldn't be started."""
errors: list[str] = []
count = 0
try:
proc = subprocess.Popen(
self._command(cookies_browser),
stdin=subprocess.DEVNULL, stdout=subprocess.PIPE,
# One stream: the markers are on stdout, the ERROR: lines
# on stderr, and one reader can't deadlock on the other.
stderr=subprocess.STDOUT,
text=True, encoding="utf-8", errors="replace", bufsize=1,
start_new_session=True)
except OSError as e:
self._busy = False
self.failed.emit(f"couldn't run yt-dlp: {e}")
return None
self._proc = proc
if self._cancel.is_set(): # cancelled before the process existed
self._terminate()
for line in proc.stdout:
parsed = parse_line(line)
if parsed is None:
if line.startswith("ERROR:"):
error = line[len("ERROR:"):].strip()
errors.append(error)
self._on_error(error)
continue
if self._on_parsed(parsed):
count += 1
proc.wait()
return count, errors, proc.returncode
class PlaylistProbe(_YtdlpRun):
"""Lists the songs at a link before anything downloads, so a playlist
— or a song link that is secretly a 900-song Mix — is shown as one
and picked from, instead of imported whole.
``entry`` streams each song as yt-dlp finds it ({id, duration, url,
title}); a long Mix takes a while to list, and the dialog fills as it
goes. ``finished`` carries how many were found (cancelled or not);
``failed`` means none were.
"""
entry = pyqtSignal(dict)
finished = pyqtSignal(int)
def __init__(self, url: str, parent=None, *,
cookies_browser: str | None = None):
super().__init__(parent, cookies_browser=cookies_browser)
self._url = url
def start(self):
if not self._busy:
self._start_thread()
def _command(self, cookies_browser):
return build_probe_command(self._url, cookies_browser)
def _on_parsed(self, parsed):
if parsed[0] != "entry" or self._cancel.is_set():
return False
self.entry.emit(parsed[1])
return True
def _run(self):
result = self._run_with_retry()
if result is None:
return
count, errors, cookies, returncode = result
self._busy = False
if count == 0 and not self._cancel.is_set():
self._fail_nothing(errors, cookies, returncode, "import")
return
self.finished.emit(count)
class UrlImportWorker(_YtdlpRun):
"""Runs yt-dlp on a daemon thread, reporting each finished file as it
lands.
``urls`` is one link, or the chosen songs' own pages after a probe.
``item_started`` is (index, count, id, title): over several links the
worker counts them itself, since each link is a playlist of one to
yt-dlp. ``item_failed`` is (id, reason) for a song yt-dlp named in an
error. ``finished`` carries {"downloaded": int, "errors": [str],
"cancelled": bool}. ``failed`` means nothing downloaded at all, and
carries yt-dlp's last error line. ``cookies_used`` names the browser
whose cookies got a bot-checked link through, so the caller can
remember it.
"""
item_started = pyqtSignal(int, int, str, str) # index, count, id, title
item_failed = pyqtSignal(str, str) # id, reason
progress = pyqtSignal(int) # percent of current item
downloaded = pyqtSignal(str) # final path of one mp3
finished = pyqtSignal(dict)
def __init__(self, urls, parent=None, *,
cookies_browser: str | None = None):
super().__init__(parent, cookies_browser=cookies_browser)
self._urls = [urls] if isinstance(urls, str) else list(urls)
self._started = 0
self.temp_dir: Path | None = None
def start(self):
if self._busy:
return
self.temp_dir = Path(tempfile.mkdtemp(prefix="lintunes-url-"))
self._start_thread()
def cleanup(self):
"""Remove our temp dir. Called by the GUI once it has imported every
file, meaning after ``finished`` or ``failed``."""
if self.temp_dir is not None:
shutil.rmtree(self.temp_dir, ignore_errors=True)
self.temp_dir = None
def _command(self, cookies_browser):
self._started = 0
return build_command(self._urls, self.temp_dir, cookies_browser)
def _on_parsed(self, parsed):
kind = parsed[0]
if kind == "start":
_, index, count, item_id, title = parsed
self._started += 1
if len(self._urls) > 1:
index, count = self._started, len(self._urls)
self.item_started.emit(index, count, item_id, title)
elif kind == "progress":
self.progress.emit(parsed[1])
elif kind == "file" and not self._cancel.is_set():
self.downloaded.emit(parsed[1])
return True
return False
def _on_error(self, error):
item = error_item(error)
if item is not None:
self.item_failed.emit(*item)
def _run(self):
result = self._run_with_retry()
if result is None:
return
count, errors, cookies, returncode = result
cancelled = self._cancel.is_set()
self._busy = False
if count == 0 and not cancelled:
self._fail_nothing(errors, cookies, returncode, "download")
return
self.finished.emit({"downloaded": count, "errors": errors,
"cancelled": cancelled})
# ---- Spotify ----
class SpotifyProbe(QObject):
"""``PlaylistProbe``'s signals for a Spotify link: reads its embed page
on a daemon thread and emits one entry per song, carrying the
``SpotifyTrack`` under ``"spotify"``. One request, no yt-dlp."""
entry = pyqtSignal(dict)
finished = pyqtSignal(int)
failed = pyqtSignal(str)
cookies_used = pyqtSignal(str) # never: Spotify needs no cookies
def __init__(self, url: str, parent=None, **_kw):
super().__init__(parent)
self._url = url
self._busy = False
self._cancel = threading.Event()
def busy(self) -> bool:
return self._busy
def cancel(self):
self._cancel.set()
def start(self):
if not self._busy:
self._busy = True
threading.Thread(target=self._run, daemon=True).start()
def _run(self):
try:
kind, source_id = spotify_link.spotify_source(self._url)
_name, tracks = spotify_link.fetch(kind, source_id)
except Exception as e:
self._busy = False
self.failed.emit(str(e))
return
count = 0
for track in tracks:
if self._cancel.is_set():
break
self.entry.emit({"id": track.id, "title": track.label(),
"duration": track.duration, "url": "",
"spotify": track})
count += 1
self._busy = False
self.finished.emit(count)
class SpotifyImportWorker(UrlImportWorker):
"""``UrlImportWorker``'s signals, for songs a Spotify link named: each
is searched for on YouTube, the best-length result downloaded (the next
one if that fails), and Spotify's tags written in before ``downloaded``.
Rows are keyed by the Spotify id, which is what the checklist and the
progress window know them by."""
# Results tried per song before giving up on it.
TRIES = 3
def __init__(self, tracks, parent=None, *,
cookies_browser: str | None = None):
super().__init__([], parent, cookies_browser=cookies_browser)
self._tracks = list(tracks)
self._stage = ""
self._query = ""
self._found: list[dict] = []
self._files: list[str] = []
def _command(self, cookies_browser):
if self._stage == "search":
return build_probe_command(
f"ytsearch{spotify_link.SEARCH_RESULTS}:{self._query}",
cookies_browser)
return build_command(self._query, self.temp_dir, cookies_browser)
def _on_parsed(self, parsed):
kind = parsed[0]
if self._stage == "search":
if kind == "entry":
self._found.append(parsed[1])
return True
elif kind == "progress":
self.progress.emit(parsed[1])
elif kind == "file" and not self._cancel.is_set():
self._files.append(parsed[1])
return True
return False
def _on_error(self, error):
pass # reported per song, by Spotify id
def _search(self, track):
"""→ (YouTube results best first, ERROR lines, cookies, code), or
None when yt-dlp couldn't run."""
self._stage, self._query, self._found = ("search",
spotify_link.search_query(
track), [])
result = self._run_with_retry()
if result is None:
return None
_count, errors, cookies, code = result
return (spotify_link.rank_candidates(self._found, track), errors,
cookies, code)
def _download(self, url):
self._stage, self._query, self._files = "download", url, []
result = self._run_with_retry()
if result is None:
return None
return self._files[-1] if self._files else "", result[1]
def _run(self):
total = len(self._tracks)
downloaded = 0
errors: list[str] = []
last = None # (ERROR lines, cookies, code) of a miss
for index, track in enumerate(self._tracks, 1):
if self._cancel.is_set():
break
self.item_started.emit(index, total, track.id, track.label())
searched = self._search(track)
if searched is None:
return
candidates, run_errors, cookies, code = searched
path, reason = "", ""
if not candidates:
reason = (run_errors[-1] if run_errors
else "no match on YouTube")
last = (run_errors, cookies, code)
for candidate in candidates[:self.TRIES]:
if self._cancel.is_set():
break
got = self._download(candidate["url"])
if got is None:
return
path, run_errors = got
if path:
break
reason = run_errors[-1] if run_errors else "download failed"
last = (run_errors, self._cookies, 0)
if self._cancel.is_set() and not path:
break
if not path:
errors.append(f"{track.label()}: {reason}")
self.item_failed.emit(track.id, reason)
continue
try:
tagging.write_tags(path, spotify_link.tag_fields(track))
except Exception as e: # untagged still beats not imported
log.warning("spotify tags for %s: %s", path, e)
downloaded += 1
self.downloaded.emit(path)
cancelled = self._cancel.is_set()
self._busy = False
if downloaded == 0 and not cancelled:
if last is not None and last[0]:
self._fail_nothing(*last, "download")
else:
self.failed.emit(errors[-1] if errors else
"nothing on the Spotify link was found "
"on YouTube")
return
self.finished.emit({"downloaded": downloaded, "errors": errors,
"cancelled": cancelled})