Files
tidal-dl/tidal_dl_ng/download.py
T

508 lines
20 KiB
Python
Raw Normal View History

2023-12-19 23:58:57 +01:00
import base64
import json
import os
import random
import shutil
import tempfile
import time
2024-01-13 11:57:11 +01:00
from collections.abc import Callable
2023-12-19 23:58:57 +01:00
from uuid import uuid4
import ffmpeg
import m3u8
import requests
2024-01-13 16:26:33 +01:00
from mpegdash.parser import MPEGDASHParser
from requests.exceptions import HTTPError
from rich.progress import Progress, TaskID
from tidalapi import Album, Mix, Playlist, Session, Track, UserPlaylist, Video
2023-12-20 07:42:36 +01:00
2023-12-19 23:58:57 +01:00
from tidal_dl_ng.config import Settings
from tidal_dl_ng.constants import (
EXTENSION_LYRICS,
REQUESTS_TIMEOUT_SEC,
AudioExtensions,
MediaType,
SkipExisting,
StreamManifestMimeType,
VideoExtensions,
)
2023-12-19 23:58:57 +01:00
from tidal_dl_ng.helper.decryption import decrypt_file, decrypt_security_token
from tidal_dl_ng.helper.exceptions import MediaMissing, UnknownManifestFormat
2024-03-29 11:16:46 +01:00
from tidal_dl_ng.helper.path import check_file_exists, format_path_media, path_file_sanitize
from tidal_dl_ng.helper.tidal import (
instantiate_media,
items_results_all,
name_builder_album_artist,
name_builder_artist,
name_builder_item,
name_builder_title,
)
2023-12-19 23:58:57 +01:00
from tidal_dl_ng.metadata import Metadata
from tidal_dl_ng.model.gui_data import ProgressBars
2024-01-14 23:27:39 +01:00
from tidal_dl_ng.model.tidal import StreamManifest
2023-12-19 23:58:57 +01:00
# TODO: Set appropriate client string and use it for video download.
# https://github.com/globocom/m3u8#using-different-http-clients
class RequestsClient:
2024-01-12 11:20:10 +01:00
def download(
self, uri: str, timeout: int = REQUESTS_TIMEOUT_SEC, headers: dict | None = None, verify_ssl: bool = True
):
2024-01-12 10:21:34 +01:00
if not headers:
headers = {}
2023-12-19 23:58:57 +01:00
o = requests.get(uri, timeout=timeout, headers=headers)
return o.text, o.url
class Download:
2024-01-23 17:22:10 +01:00
settings: Settings
session: Session
2024-01-13 11:57:11 +01:00
skip_existing: SkipExisting = False
2024-01-23 17:22:10 +01:00
fn_logger: Callable
progress_gui: ProgressBars
progress: Progress
2023-12-19 23:58:57 +01:00
2024-01-19 08:17:47 +01:00
def __init__(
self,
session: Session,
path_base: str,
fn_logger: Callable,
skip_existing: SkipExisting = SkipExisting.Disabled,
progress_gui: ProgressBars = None,
progress: Progress = None,
):
2024-01-14 18:10:58 +01:00
self.settings = Settings()
2023-12-19 23:58:57 +01:00
self.session = session
self.skip_existing = skip_existing
2024-01-19 08:17:47 +01:00
self.fn_logger = fn_logger
self.progress_gui = progress_gui
self.progress = progress
self.path_base = path_base
2023-12-19 23:58:57 +01:00
2024-03-29 11:16:46 +01:00
if not self.settings.data.path_binary_ffmpeg and self.settings.data.video_convert_mp4:
self.settings.data.video_convert_mp4 = False
self.fn_logger.error(
2024-03-29 11:16:46 +01:00
"FFmpeg in path is not set. Videos can be downloaded but will not be processed. "
"Make sure FFmpeg is installed and the path to the binary is configured ('path_binary_ffmpeg')."
)
def _download(
2024-01-13 16:26:33 +01:00
self,
media: Track | Video,
2024-01-13 16:26:33 +01:00
stream_manifest: StreamManifest,
path_file: str,
2024-01-19 08:17:47 +01:00
) -> str:
2024-01-13 16:26:33 +01:00
media_name: str = name_builder_item(media)
# Set the correct progress output channel.
2024-01-19 08:17:47 +01:00
if self.progress_gui is None:
2024-01-13 16:26:33 +01:00
progress_stdout: bool = True
else:
progress_stdout: bool = False
# Send signal to GUI with media name
2024-01-19 08:17:47 +01:00
self.progress_gui.item_name.emit(media_name[:30])
2024-01-13 16:26:33 +01:00
try:
# Compute total iterations for progress
urls_count: int = len(stream_manifest.urls)
2024-01-13 16:26:33 +01:00
if urls_count > 1:
progress_total: int = urls_count
block_size: int | None = None
else:
# Compute progress iterations based on the file size.
r = requests.get(stream_manifest.urls[0], stream=True, timeout=REQUESTS_TIMEOUT_SEC)
2024-01-13 16:26:33 +01:00
r.raise_for_status()
# Get file size and compute progress steps
total_size_in_bytes: int = int(r.headers.get("content-length", 0))
block_size: int | None = 4096
progress_total: float = total_size_in_bytes / block_size
# Create progress Task
2024-01-19 08:17:47 +01:00
p_task: TaskID = self.progress.add_task(
2024-01-15 21:22:34 +01:00
f"[blue]Item '{media_name[:30]}'",
total=progress_total,
2024-01-13 16:26:33 +01:00
visible=progress_stdout,
)
# Write content to file until progress is finished.
2024-01-19 08:17:47 +01:00
while not self.progress.tasks[p_task].finished:
2024-01-13 16:26:33 +01:00
with open(path_file, "wb") as f:
for url in stream_manifest.urls:
# Create the request object with stream=True, so the content won't be loaded into memory at once.
r = requests.get(url, stream=True, timeout=REQUESTS_TIMEOUT_SEC)
2024-01-13 16:26:33 +01:00
r.raise_for_status()
# Write the content to disk. If `chunk_size` is set to `None` the whole file will be written at once.
for data in r.iter_content(chunk_size=block_size):
f.write(data)
# Advance progress bar.
2024-01-19 08:17:47 +01:00
self.progress.advance(p_task)
# To send the progress to the GUI, we need to emit the percentage.
if not progress_stdout:
2024-01-19 08:17:47 +01:00
self.progress_gui.item.emit(self.progress.tasks[p_task].percentage)
2024-01-13 16:26:33 +01:00
except HTTPError as e:
# TODO: Handle Exception...
2024-01-19 08:17:47 +01:00
self.fn_logger(e)
2024-01-13 16:26:33 +01:00
# Check if file is encrypted.
needs_decryption = self.is_encrypted(stream_manifest.encryption_type)
if needs_decryption:
key, nonce = decrypt_security_token(stream_manifest.encryption_key)
tmp_path_file_decrypted = path_file + "_decrypted"
decrypt_file(path_file, tmp_path_file_decrypted, key, nonce)
else:
tmp_path_file_decrypted = path_file
# Write metadata to file.
if not isinstance(media, Video):
self.metadata_write(media, tmp_path_file_decrypted)
2024-01-13 16:26:33 +01:00
return tmp_path_file_decrypted
2023-12-19 23:58:57 +01:00
def item(
self,
2024-01-13 11:57:11 +01:00
file_template: str,
2023-12-19 23:58:57 +01:00
media: Track | Video = None,
2024-01-13 11:57:11 +01:00
media_id: str = None,
2023-12-19 23:58:57 +01:00
media_type: MediaType = None,
video_download: bool = True,
download_delay: bool = False,
2023-12-19 23:58:57 +01:00
) -> (bool, str):
try:
if media_id and media_type:
# If no media instance is provided, we need to create the media instance.
media = instantiate_media(self.session, media_type, media_id)
elif isinstance(media, Track): # Check if media is available not deactivated / removed from TIDAL.
if not media.available:
self.fn_logger.info(
f"This track is not available for listening anymore on TIDAL. Skipping: {name_builder_item(media)}"
)
else:
# Re-create media instance with full album information
media = self.session.track(media.id, with_album=True)
elif not media:
raise MediaMissing
except:
2024-02-26 06:53:53 +01:00
return False, ""
2024-01-13 11:57:11 +01:00
# If video download is not allowed end here
2024-02-26 06:48:11 +01:00
if not video_download and isinstance(media, Video):
2024-01-19 08:17:47 +01:00
self.fn_logger.info(
2024-01-13 11:57:11 +01:00
f"Video downloads are deactivated (see settings). Skipping video: {name_builder_item(media)}"
)
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
return False, ""
2023-12-19 23:58:57 +01:00
2024-01-13 16:26:33 +01:00
# Populate StreamManifest for further download.
2023-12-19 23:58:57 +01:00
if isinstance(media, Track):
stream = media.get_stream()
2024-01-13 16:26:33 +01:00
manifest: str = stream.manifest
mime_type: str = stream.manifest_mime_type
else:
manifest: str = media.get_url()
mime_type: str = StreamManifestMimeType.VIDEO.value
stream_manifest = self.stream_manifest_parse(manifest, mime_type)
2023-12-19 23:58:57 +01:00
# Create file name and path
file_name_relative = format_path_media(file_template, media)
path_file = os.path.abspath(
os.path.normpath(os.path.join(os.path.expanduser(self.path_base), file_name_relative))
)
2024-01-13 11:57:11 +01:00
# Sanitize final path_file to fit into OS boundaries.
uniquify: bool = self.skip_existing == SkipExisting.Append
path_file = path_file_sanitize(path_file + stream_manifest.file_extension, adapt=True, uniquify=uniquify)
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
# Compute if and how downloads need to be skipped.
if self.skip_existing.value in (SkipExisting.ExtensionIgnore.value, SkipExisting.Filename.value):
extension_ignore: bool = self.skip_existing == SkipExisting.ExtensionIgnore
download_skip: bool = check_file_exists(path_file, extension_ignore=extension_ignore)
2024-01-13 11:57:11 +01:00
else:
download_skip: bool = False
2023-12-19 23:58:57 +01:00
if not download_skip:
2024-01-13 11:57:11 +01:00
# Create a temp directory and file.
2023-12-19 23:58:57 +01:00
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmp_path_dir:
tmp_path_file = os.path.join(tmp_path_dir, str(uuid4()) + stream_manifest.file_extension)
# Download media.
2024-01-19 08:17:47 +01:00
tmp_path_file = self._download(media=media, stream_manifest=stream_manifest, path_file=tmp_path_file)
2023-12-19 23:58:57 +01:00
2024-01-14 18:10:58 +01:00
if isinstance(media, Video) and self.settings.data.video_convert_mp4:
2024-01-13 11:57:11 +01:00
# Convert `*.ts` file to `*.mp4` using ffmpeg
tmp_path_file = self._video_convert(tmp_path_file)
path_file = os.path.splitext(path_file)[0] + ".mp4"
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
# Move final file to the configured destination directory.
2023-12-19 23:58:57 +01:00
os.makedirs(os.path.dirname(path_file), exist_ok=True)
2024-01-13 11:57:11 +01:00
shutil.move(tmp_path_file, path_file)
# Move lyrics file
if self.settings.data.lyrics_file:
self._move_lyrics(path_file, tmp_path_file)
2023-12-19 23:58:57 +01:00
else:
2024-01-19 08:17:47 +01:00
self.fn_logger.debug(f"Download skipped, since file exists: '{path_file}'")
2023-12-19 23:58:57 +01:00
status_download: bool = not download_skip
# If a file was downloaded and the download delay is enabled, wait until the next download.
# Only use this, if you have a list of several Track items. Do not use this for list items.
if download_delay and status_download:
time_sleep: float = round(random.SystemRandom().uniform(2, 5), 1)
self.fn_logger.debug(f"Next download will start in {time_sleep} seconds.")
time.sleep(time_sleep)
2023-12-19 23:58:57 +01:00
return not download_skip, path_file
def _move_lyrics(self, file_media_dst: str, file_media_src: str):
# Build tmp lyrics filename
tmp_lyrics_file_path: str = file_media_src + EXTENSION_LYRICS
# Check if the file was downloaded
if os.path.isfile(tmp_lyrics_file_path):
# Move it.
shutil.move(tmp_lyrics_file_path, os.path.splitext(file_media_dst)[0] + EXTENSION_LYRICS)
def lyrics_write_file(self, file_path: str, lyrics: str) -> str:
result: str = file_path
try:
with open(file_path, "x", encoding="utf-8") as f:
f.write(lyrics)
except:
result = ""
return result
2023-12-19 23:58:57 +01:00
def metadata_write(self, track: Track, path_file: str):
result: bool = False
release_date: str = (
2024-03-22 15:02:26 +01:00
track.album.available_release_date.strftime("%Y-%m-%d")
if track.album.available_release_date
else track.album.release_date.strftime("%Y-%m-%d") if track.album.release_date else ""
)
copy_right: str = track.copyright if hasattr(track, "copyright") and track.copyright else ""
isrc: str = track.isrc if hasattr(track, "isrc") and track.isrc else ""
2024-01-14 18:04:31 +01:00
lyrics: str = ""
2023-12-19 23:58:57 +01:00
if self.settings.data.lyrics_embed or self.settings.data.lyrics_file:
2024-01-14 18:04:31 +01:00
# Try to retrieve lyrics.
try:
lyrics_obj = track.lyrics()
if lyrics_obj.subtitles:
lyrics = lyrics_obj.subtitles
elif lyrics_obj.text:
lyrics = lyrics_obj.text
2024-02-09 11:01:25 +01:00
except (HTTPError, AttributeError):
lyrics = ""
2024-01-14 18:04:31 +01:00
# TODO: Implement proper logging.
print(f"Could not retrieve lyrics for `{name_builder_item(track)}`.")
2023-12-19 23:58:57 +01:00
2024-02-09 11:01:25 +01:00
if lyrics and self.settings.data.lyrics_file:
self.lyrics_write_file(path_file + EXTENSION_LYRICS, lyrics)
# `None` values are not allowed.
2023-12-19 23:58:57 +01:00
m: Metadata = Metadata(
path_file=path_file,
lyrics=lyrics,
copy_right=copy_right,
2024-03-14 06:30:47 +01:00
title=name_builder_title(track),
artists=name_builder_artist(track),
album=track.album.name if track.album else "",
2023-12-19 23:58:57 +01:00
tracknumber=track.track_num,
date=release_date,
2023-12-20 07:42:36 +01:00
isrc=isrc,
albumartist=name_builder_album_artist(track),
totaltrack=track.album.num_tracks if track.album and track.album.num_tracks else 1,
totaldisc=track.album.num_volumes if track.album and track.album.num_volumes else 1,
discnumber=track.volume_num if track.volume_num else 1,
url_cover=track.album.image(self.settings.data.metadata_cover_dimension.value),
2023-12-19 23:58:57 +01:00
)
m.save()
result = True
return result
2024-01-12 10:21:34 +01:00
def items(
2023-12-19 23:58:57 +01:00
self,
2024-01-19 08:17:47 +01:00
file_template: str,
media: Album | Playlist | UserPlaylist | Mix = None,
2024-01-14 17:58:54 +01:00
media_id: str = None,
2023-12-19 23:58:57 +01:00
media_type: MediaType = None,
video_download: bool = False,
download_delay: bool = True,
):
2024-01-14 17:58:54 +01:00
# If no media instance is provided, we need to create the media instance.
if media_id and media_type:
media = instantiate_media(self.session, media_type, media_id)
2024-01-14 17:58:54 +01:00
elif not media:
raise MediaMissing
2023-12-19 23:58:57 +01:00
2024-01-14 17:58:54 +01:00
# Create file name and path
file_name_relative = format_path_media(file_template, media)
2023-12-19 23:58:57 +01:00
2024-01-18 22:14:46 +01:00
# Get the name of the list and check, if videos should be included.
videos_include: bool = True
2024-01-14 17:58:54 +01:00
if isinstance(media, Mix):
2024-01-15 21:22:34 +01:00
list_media_name = media.title[:30]
2023-12-19 23:58:57 +01:00
elif video_download:
2024-03-14 06:30:47 +01:00
list_media_name = name_builder_title(media)[:30]
2023-12-19 23:58:57 +01:00
else:
2024-01-18 22:14:46 +01:00
videos_include = False
2024-03-14 06:30:47 +01:00
list_media_name = name_builder_title(media)[:30]
2023-12-19 23:58:57 +01:00
2024-01-18 22:14:46 +01:00
# Get all items of the list.
items = items_results_all(media, videos_include=videos_include)
2024-01-14 17:58:54 +01:00
# Determine where to redirect the progress information.
2024-01-19 08:17:47 +01:00
if self.progress_gui is None:
2023-12-19 23:58:57 +01:00
progress_stdout: bool = True
else:
progress_stdout: bool = False
2024-01-19 08:17:47 +01:00
self.progress_gui.item_name.emit(list_media_name[:30])
2023-12-19 23:58:57 +01:00
2024-01-14 17:58:54 +01:00
# Create the list progress task.
2024-01-19 08:17:47 +01:00
p_task1: TaskID = self.progress.add_task(
2024-01-14 17:58:54 +01:00
f"[green]List '{list_media_name}'", total=len(items), visible=progress_stdout
)
2023-12-19 23:58:57 +01:00
2024-01-14 17:58:54 +01:00
# Iterate through list items
2024-01-19 08:17:47 +01:00
while not self.progress.finished:
2023-12-19 23:58:57 +01:00
for media in items:
2024-01-14 17:58:54 +01:00
# Download the item.
status_download, result_path_file = self.item(
2023-12-19 23:58:57 +01:00
media=media,
2024-01-19 08:17:47 +01:00
file_template=file_name_relative,
2023-12-19 23:58:57 +01:00
)
2024-01-14 17:58:54 +01:00
# Advance progress bar.
2024-01-19 08:17:47 +01:00
self.progress.advance(p_task1)
2023-12-19 23:58:57 +01:00
if not progress_stdout:
2024-01-19 08:17:47 +01:00
self.progress_gui.list_item.emit(self.progress.tasks[p_task1].percentage)
2023-12-19 23:58:57 +01:00
2024-01-14 17:58:54 +01:00
# If a file was downloaded and the download delay is enabled, wait until the next download.
if download_delay and status_download:
2023-12-19 23:58:57 +01:00
time_sleep: float = round(random.SystemRandom().uniform(2, 5), 1)
2024-01-19 08:17:47 +01:00
self.fn_logger.debug(f"Next download will start in {time_sleep} seconds.")
2023-12-19 23:58:57 +01:00
time.sleep(time_sleep)
2024-01-13 11:57:11 +01:00
def is_encrypted(self, encryption_type: str) -> bool:
result = encryption_type != "NONE"
2023-12-19 23:58:57 +01:00
return result
def get_file_extension(self, stream_url: str, stream_codec: str) -> str:
if AudioExtensions.FLAC.value in stream_url:
result: str = AudioExtensions.FLAC.value
elif AudioExtensions.MP4.value in stream_url:
2024-01-27 17:27:57 +01:00
if "ac4" in stream_codec or "mha1" in stream_codec or "flac" in stream_codec or "mp4a" in stream_codec:
result: str = AudioExtensions.M4A.value
else:
result: str = AudioExtensions.MP4.value
elif VideoExtensions.TS.value in stream_url:
result: str = VideoExtensions.TS.value
2023-12-19 23:58:57 +01:00
else:
2024-01-27 17:27:57 +01:00
result: str = AudioExtensions.MP4.value
2023-12-19 23:58:57 +01:00
return result
def _video_convert(self, path_file: str) -> str:
path_file_out = os.path.splitext(path_file)[0] + AudioExtensions.MP4.value
2023-12-19 23:58:57 +01:00
result, _ = ffmpeg.input(path_file).output(path_file_out, map=0, c="copy").run()
return path_file_out
2024-01-13 11:57:11 +01:00
2024-01-13 16:26:33 +01:00
def stream_manifest_parse(self, manifest: str, mime_type: str) -> StreamManifest:
if mime_type == StreamManifestMimeType.MPD.value:
# Stream Manifest is base64 encoded.
manifest_parsed: str = base64.b64decode(manifest).decode("utf-8")
mpd = MPEGDASHParser.parse(manifest_parsed)
codecs: str = mpd.periods[0].adaptation_sets[0].representations[0].codecs
mime_type: str = mpd.periods[0].adaptation_sets[0].mime_type
2024-01-13 11:57:11 +01:00
# TODO: Handle encryption key. But I have never seen an encrypted file so far.
encryption_type: str = "NONE"
encryption_key: str | None = None
2024-01-13 16:26:33 +01:00
# .initialization + the very first of .media; See https://developers.broadpeak.io/docs/foundations-dash
segments_count = 1 + 1
for s in mpd.periods[0].adaptation_sets[0].representations[0].segment_templates[0].segment_timelines[0].Ss:
segments_count += s.r if s.r else 1
# Populate segment urls.
segment_template = mpd.periods[0].adaptation_sets[0].representations[0].segment_templates[0]
stream_urls: list[str] = []
for index in range(segments_count):
stream_urls.append(segment_template.media.replace("$Number$", str(index)))
2024-01-13 16:26:33 +01:00
elif mime_type == StreamManifestMimeType.BTS.value:
2024-01-13 16:26:33 +01:00
# Stream Manifest is base64 encoded.
manifest_parsed: str = base64.b64decode(manifest).decode("utf-8")
2024-01-13 11:57:11 +01:00
# JSON string to object.
stream_manifest = json.loads(manifest_parsed)
# TODO: Handle more than one download URL
2024-01-13 16:26:33 +01:00
stream_urls: str = stream_manifest["urls"]
2024-01-13 11:57:11 +01:00
codecs: str = stream_manifest["codecs"]
mime_type: str = stream_manifest["mimeType"]
encryption_type: str = stream_manifest["encryptionType"]
encryption_key: str | None = (
stream_manifest["encryptionKey"] if self.is_encrypted(encryption_type) else None
)
2024-01-13 16:26:33 +01:00
elif mime_type == StreamManifestMimeType.VIDEO.value:
# Parse M3U8 video playlist
m3u8_variant: m3u8.M3U8 = m3u8.load(manifest)
# Find the desired video resolution or the next best one.
2024-01-14 18:10:58 +01:00
m3u8_playlist, codecs = self._extract_video_stream(m3u8_variant, self.settings.data.quality_video.value)
2024-01-13 16:26:33 +01:00
# Populate urls.
stream_urls: list[str] = m3u8_playlist.files
# TODO: Handle encryption key. But I have never seen an encrypted file so far.
encryption_type: str = "NONE"
encryption_key: str | None = None
2024-01-13 11:57:11 +01:00
else:
raise UnknownManifestFormat
2024-01-13 16:26:33 +01:00
file_extension: str = self.get_file_extension(stream_urls[0], codecs)
2024-01-13 11:57:11 +01:00
result: StreamManifest = StreamManifest(
urls=stream_urls,
2024-01-13 11:57:11 +01:00
codecs=codecs,
file_extension=file_extension,
encryption_type=encryption_type,
encryption_key=encryption_key,
mime_type=mime_type,
2024-01-13 11:57:11 +01:00
)
return result
2024-01-13 16:26:33 +01:00
def _extract_video_stream(self, m3u8_variant: m3u8.M3U8, quality: str) -> (m3u8.M3U8 | bool, str):
m3u8_playlist: m3u8.M3U8 | bool = False
resolution_best: int = 0
mime_type: str = ""
if m3u8_variant.is_variant:
for playlist in m3u8_variant.playlists:
if resolution_best < playlist.stream_info.resolution[1]:
resolution_best = playlist.stream_info.resolution[1]
m3u8_playlist = m3u8.load(playlist.uri)
mime_type = playlist.stream_info.codecs
if quality == playlist.stream_info.resolution[1]:
break
return m3u8_playlist, mime_type