Files
tidal-dl/tidal_dl_ng/download.py
T

501 lines
19 KiB
Python
Raw Normal View History

2023-12-19 23:58:57 +01:00
import base64
import json
import os
import random
import shutil
import tempfile
import time
2024-01-13 11:57:11 +01:00
from collections.abc import Callable
2023-12-19 23:58:57 +01:00
from logging import Logger
from uuid import uuid4
import ffmpeg
import m3u8
import requests
2024-01-13 11:57:11 +01:00
from helper.tidal import name_builder_item
from model.tidal import StreamManifest
2023-12-20 07:42:36 +01:00
from requests.exceptions import HTTPError
from rich.progress import Progress
from tidalapi import Album, Mix, Playlist, Session, Track, UserPlaylist, Video
2024-01-13 16:26:33 +01:00
from mpegdash.parser import MPEGDASHParser
2023-12-20 07:42:36 +01:00
2023-12-19 23:58:57 +01:00
from tidal_dl_ng.config import Settings
2024-01-13 16:26:33 +01:00
from tidal_dl_ng.constants import REQUESTS_TIMEOUT_SEC, MediaType, SkipExisting, StreamManifestMimeType
2023-12-19 23:58:57 +01:00
from tidal_dl_ng.helper.decryption import decrypt_file, decrypt_security_token
2024-01-13 11:57:11 +01:00
from tidal_dl_ng.helper.exceptions import MediaMissing, MediaUnknown, UnknownManifestFormat
from tidal_dl_ng.helper.path import check_file_exists, format_path_media, path_file_sanitize
2023-12-19 23:58:57 +01:00
from tidal_dl_ng.helper.wrapper import WrapperLogger
from tidal_dl_ng.metadata import Metadata
from tidal_dl_ng.model.gui_data import ProgressBars
# TODO: Set appropriate client string and use it for video download.
# https://github.com/globocom/m3u8#using-different-http-clients
class RequestsClient:
2024-01-12 11:20:10 +01:00
def download(
self, uri: str, timeout: int = REQUESTS_TIMEOUT_SEC, headers: dict | None = None, verify_ssl: bool = True
):
2024-01-12 10:21:34 +01:00
if not headers:
headers = {}
2023-12-19 23:58:57 +01:00
o = requests.get(uri, timeout=timeout, headers=headers)
return o.text, o.url
class Download:
# TODO: Implement download cover 1280.
session: Session = None
2024-01-13 11:57:11 +01:00
skip_existing: SkipExisting = False
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
def __init__(self, session: Session, skip_existing: SkipExisting = SkipExisting.Disabled):
2023-12-19 23:58:57 +01:00
self.session = session
self.skip_existing = skip_existing
2024-01-13 16:26:33 +01:00
def _audio_stream(
self,
fn_logger: Callable,
media: Track,
progress: Progress,
progress_gui: ProgressBars,
stream_manifest: StreamManifest,
path_file: str,
):
media_name: str = name_builder_item(media)
# Set the correct progress output channel.
if progress_gui is None:
progress_stdout: bool = True
else:
progress_stdout: bool = False
progress_gui.item_name.emit(media_name)
try:
# Download the media as stream, so we can iterate over the response.
r = requests.get(stream_manifest.stream_urls, stream=True, timeout=REQUESTS_TIMEOUT_SEC)
r.raise_for_status()
# Get file size and compute progress steps
total_size_in_bytes = int(r.headers.get("content-length", 0))
block_size = 4096
p_task = progress.add_task(
f"[blue]Item '{media_name[:30]}'",
total=total_size_in_bytes / block_size,
visible=progress_stdout,
)
# Write content to file until progress is finished.
while not progress.tasks[p_task].finished:
with open(path_file, "wb") as f:
for data in r.iter_content(chunk_size=block_size):
f.write(data)
# Advance progress bar.
progress.advance(p_task)
# To send the progress to the GUI, we need to emit the percentage.
if not progress_stdout:
progress_gui.item.emit(progress.tasks[p_task].percentage)
except HTTPError as e:
# TODO: Handle Exception...
fn_logger(e)
# Check if file is encrypted.
needs_decryption = self.is_encrypted(stream_manifest.encryption_type)
if needs_decryption:
key, nonce = decrypt_security_token(stream_manifest.encryption_key)
tmp_path_file_decrypted = path_file + "_decrypted"
decrypt_file(path_file, tmp_path_file_decrypted, key, nonce)
else:
tmp_path_file_decrypted = path_file
# Write metadata to file.
self.metadata_write(media, tmp_path_file_decrypted)
return tmp_path_file_decrypted
def _mpeg_segments(
self,
fn_logger: Callable,
media: Track,
progress: Progress,
progress_gui: ProgressBars,
stream_manifest: StreamManifest,
path_file: str,
):
media_name: str = name_builder_item(media)
# Set the correct progress output channel.
if progress_gui is None:
progress_stdout: bool = True
else:
progress_stdout: bool = False
progress_gui.item_name.emit(media_name)
try:
total_iterations = stream_manifest.segments_count
p_task = progress.add_task(
f"[blue]Item '{media.name[:30]}'",
total=total_iterations,
visible=progress_stdout,
)
# Write content to file until progress is finished.
while not progress.tasks[p_task].finished:
with open(path_file, "wb") as f:
for index in range(total_iterations):
# Download the media.
segment_url = stream_manifest.stream_urls.replace('$Number$', str(index))
r = requests.get(segment_url, timeout=REQUESTS_TIMEOUT_SEC)
r.raise_for_status()
# Write data
f.write(r.content)
# Advance progress bar.
progress.advance(p_task)
# To send the progress to the GUI, we need to emit the percentage.
if not progress_stdout:
progress_gui.item.emit(progress.tasks[p_task].percentage)
except HTTPError as e:
# TODO: Handle Exception...
fn_logger(e)
return path_file
2024-01-13 11:57:11 +01:00
2023-12-19 23:58:57 +01:00
def _video(self, video: Video, path_file: str) -> str | None:
result: str | None = None
2024-01-13 16:26:33 +01:00
with open(path_file, "wb") as f:
for segment in m3u8_playlist.data["segments"]:
url = segment["uri"]
r = requests.get(url, timeout=REQUESTS_TIMEOUT_SEC)
2023-12-19 23:58:57 +01:00
2024-01-13 16:26:33 +01:00
f.write(r.content)
2023-12-19 23:58:57 +01:00
2024-01-13 16:26:33 +01:00
result = path_file
2023-12-19 23:58:57 +01:00
return result
2024-01-13 11:57:11 +01:00
def instantiate_media(
2024-01-13 16:26:33 +01:00
self, session: Session, media_type: type[MediaType.Track, MediaType.Video], id_media: str
2024-01-13 11:57:11 +01:00
) -> Track | Video:
if media_type == MediaType.Track:
media = Track(session, id_media)
elif media_type == MediaType.Video:
media = Video(session, id_media)
else:
raise MediaUnknown
return media
2023-12-19 23:58:57 +01:00
def item(
self,
path_base: str,
2024-01-13 11:57:11 +01:00
file_template: str,
fn_logger: Callable,
2023-12-19 23:58:57 +01:00
media: Track | Video = None,
2024-01-13 11:57:11 +01:00
media_id: str = None,
2023-12-19 23:58:57 +01:00
media_type: MediaType = None,
video_download: bool = True,
progress_gui: ProgressBars = None,
2023-12-20 07:42:36 +01:00
progress: Progress = None,
2023-12-19 23:58:57 +01:00
) -> (bool, str):
2024-01-13 11:57:11 +01:00
# If only a media_id is provided, we need to create the media instance.
if media_id and media_type:
media = self.instantiate_media(self.session, media_type, media_id)
elif not media:
raise MediaMissing
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
# If video download is not allowed end here
if not video_download:
fn_logger.info(
f"Video downloads are deactivated (see settings). Skipping video: {name_builder_item(media)}"
)
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
return False, ""
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
# Create file name and path
file_name_relative = format_path_media(file_template, media)
path_file = os.path.abspath(os.path.normpath(os.path.join(path_base, file_name_relative)))
2024-01-13 16:26:33 +01:00
# Populate StreamManifest for further download.
2023-12-19 23:58:57 +01:00
if isinstance(media, Track):
stream = media.stream()
2024-01-13 16:26:33 +01:00
manifest: str = stream.manifest
mime_type: str = stream.manifest_mime_type
else:
manifest: str = media.get_url()
mime_type: str = StreamManifestMimeType.VIDEO.value
stream_manifest = self.stream_manifest_parse(manifest, mime_type)
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
# Sanitize final path_file to fit into OS boundaries.
2024-01-13 16:26:33 +01:00
path_file = path_file_sanitize(path_file + stream_manifest.file_extension, adapt=True)
2023-12-19 23:58:57 +01:00
2024-01-13 11:57:11 +01:00
# Compute if and how downloads need to be skipped.
2024-01-13 16:26:33 +01:00
if self.skip_existing.value:
2024-01-13 12:04:35 +01:00
extension_ignore = self.skip_existing == SkipExisting.ExtensionIgnore
2024-01-13 11:57:11 +01:00
# TODO: Check if extension is already in `path_file` or not.
download_skip = check_file_exists(path_file, extension_ignore=extension_ignore)
else:
download_skip = False
2023-12-19 23:58:57 +01:00
if not download_skip:
2024-01-13 11:57:11 +01:00
# Create a temp directory and file.
2023-12-19 23:58:57 +01:00
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmp_path_dir:
2024-01-13 11:57:11 +01:00
tmp_path_file = os.path.join(tmp_path_dir, str(uuid4()))
2023-12-19 23:58:57 +01:00
if isinstance(media, Track):
2024-01-13 16:26:33 +01:00
if stream_manifest.segments_count > 0:
tmp_path_file = self._mpeg_segments(fn_logger, media, progress, progress_gui, stream_manifest, tmp_path_file)
else:
tmp_path_file = self._audio_stream(
fn_logger, media, progress, progress_gui, stream_manifest, tmp_path_file
)
2023-12-19 23:58:57 +01:00
elif isinstance(media, Video):
2024-01-13 11:57:11 +01:00
tmp_path_file = self._video(media, tmp_path_file)
2023-12-19 23:58:57 +01:00
# TODO: Check if is possible to write metadata to MPEG Transport Stream files.
# TODO: Make optional.
2024-01-13 11:57:11 +01:00
# Convert `*.ts` file to `*.mp4` using ffmpeg
2023-12-19 23:58:57 +01:00
if True:
2024-01-13 11:57:11 +01:00
tmp_path_file = self._video_convert(tmp_path_file)
2023-12-19 23:58:57 +01:00
path_file = os.path.splitext(path_file)[0] + ".mp4"
2024-01-13 11:57:11 +01:00
# Move final file to the configured destination directory.
2023-12-19 23:58:57 +01:00
os.makedirs(os.path.dirname(path_file), exist_ok=True)
2024-01-13 11:57:11 +01:00
shutil.move(tmp_path_file, path_file)
2023-12-19 23:58:57 +01:00
else:
fn_logger.debug(f"Download skipped, since file exists: '{path_file}'")
return not download_skip, path_file
def cover_url(self, sid: str, width: int = 320, height: int = 320):
if sid is None:
return ""
return f"https://resources.tidal.com/images/{sid.replace('-', '/')}/{int(width)}x{int(height)}.jpg"
def metadata_write(self, track: Track, path_file: str):
settings: Settings = Settings()
2023-12-19 23:58:57 +01:00
result: bool = False
release_date: str = track.album.release_date.strftime("%Y-%m-%d") if track.album.release_date else ""
copy_right: str = track.copyright if track.copyright else ""
2023-12-20 07:42:36 +01:00
isrc: str = track.isrc if track.isrc else ""
2023-12-19 23:58:57 +01:00
try:
2023-12-31 21:23:36 +01:00
lyrics: str = track.lyrics().subtitles if hasattr(track, "lyrics") else ""
2023-12-19 23:58:57 +01:00
except HTTPError:
lyrics: str = ""
2023-12-20 07:42:36 +01:00
# TODO: Check if it is possible to pass "None" values.
2023-12-19 23:58:57 +01:00
m: Metadata = Metadata(
path_file=path_file,
lyrics=lyrics,
copy_right=copy_right,
title=track.name,
artists=[artist.name for artist in track.artists],
album=track.album.name,
tracknumber=track.track_num,
date=release_date,
2023-12-20 07:42:36 +01:00
isrc=isrc,
2023-12-19 23:58:57 +01:00
albumartist=track.artist.name,
totaltrack=track.album.num_tracks if track.album.num_tracks else 1,
totaldisc=track.album.num_volumes if track.album.num_volumes else 1,
discnumber=track.volume_num,
2024-01-12 10:51:30 +01:00
url_cover=self.cover_url(
track.album.cover, settings.data.metadata_cover_width, settings.data.metadata_cover_height
),
2023-12-19 23:58:57 +01:00
)
m.save()
result = True
return result
2024-01-12 10:21:34 +01:00
def items(
2023-12-19 23:58:57 +01:00
self,
path_base: str,
fn_logger: Logger | WrapperLogger,
id_media: str = None,
media_type: MediaType = None,
file_template: str = None,
list_media: Album | Playlist | UserPlaylist | Mix = None,
video_download: bool = False,
progress_gui: ProgressBars = None,
progress: Progress = None,
download_delay: bool = True,
):
if not list_media:
if media_type == MediaType.Album:
list_media = Album(self.session, id_media)
elif media_type == MediaType.Playlist:
list_media = Playlist(self.session, id_media)
elif media_type == MediaType.Mix:
list_media = Mix(self.session, id_media)
else:
raise MediaUnknown
if file_template:
file_name_relative = format_path_media(file_template, list_media)
path_file = path_base
else:
file_name_relative = file_template
path_file = format_path_media(path_base, list_media)
# TODO: Extend with pagination support: Iterate through `items` and `tracks`until len(returned list) == 0
if isinstance(list_media, Mix):
items = list_media.items()
list_media_name = list_media.title[:30]
elif video_download:
items = list_media.items(limit=100)
list_media_name = list_media.name[:30]
else:
items = list_media.tracks(limit=999)
list_media_name = list_media.name[:30]
if progress_gui is None:
progress_stdout: bool = True
else:
progress_stdout: bool = False
p_task1 = progress.add_task(f"[green]List '{list_media_name}'", total=len(items), visible=progress_stdout)
while not progress.finished:
for media in items:
Progress()
# TODO: Handle return value of `track` method.
status_download, result_path_file = self.item(
2023-12-19 23:58:57 +01:00
path_base=path_file,
file_template=file_name_relative,
media=media,
progress_gui=progress_gui,
progress=progress,
2023-12-20 07:42:36 +01:00
fn_logger=fn_logger,
2023-12-19 23:58:57 +01:00
)
progress.advance(p_task1)
if not progress_stdout:
progress_gui.list_item.emit(progress.tasks[p_task1].percentage)
if download_delay and status_download:
2023-12-19 23:58:57 +01:00
time_sleep: float = round(random.SystemRandom().uniform(2, 5), 1)
# TODO: Fix logging. Is not displayed in debug window.
fn_logger.debug(f"Next download will start in {time_sleep} seconds.")
time.sleep(time_sleep)
2024-01-13 11:57:11 +01:00
def is_encrypted(self, encryption_type: str) -> bool:
result = encryption_type != "NONE"
2023-12-19 23:58:57 +01:00
return result
def get_file_extension(self, stream_url: str, stream_codec: str) -> str:
if ".flac" in stream_url:
2024-01-13 16:26:33 +01:00
result: str = ".flac"
2023-12-19 23:58:57 +01:00
elif ".mp4" in stream_url:
2024-01-13 16:26:33 +01:00
# TODO: Need to investigate, what the correct extension is.
# if "ac4" in stream_codec or "mha1" in stream_codec:
# result = ".mp4"
# elif "flac" in stream_codec:
# result = ".flac"
# else:
# result = ".m4a"
result: str = ".mp4"
if ".ts" in stream_url:
result: str = ".ts"
2023-12-19 23:58:57 +01:00
else:
2024-01-13 16:26:33 +01:00
result: str = ".m4a"
2023-12-19 23:58:57 +01:00
return result
def _video_convert(self, path_file: str) -> str:
path_file_out = os.path.splitext(path_file)[0] + ".mp4"
result, _ = ffmpeg.input(path_file).output(path_file_out, map=0, c="copy").run()
return path_file_out
2024-01-13 11:57:11 +01:00
2024-01-13 16:26:33 +01:00
def stream_manifest_parse(self, manifest: str, mime_type: str) -> StreamManifest:
if mime_type == StreamManifestMimeType.MPD.value:
# Stream Manifest is base64 encoded.
manifest_parsed: str = base64.b64decode(manifest).decode("utf-8")
mpd = MPEGDASHParser.parse(manifest_parsed)
codecs: str = mpd.periods[0].adaptation_sets[0].representations[0].codecs
mime_type: str = mpd.periods[0].adaptation_sets[0].mime_type
2024-01-13 11:57:11 +01:00
# TODO: Handle encryption key. But I have never seen an encrypted file so far.
encryption_type: str = "NONE"
encryption_key: str | None = None
2024-01-13 16:26:33 +01:00
# .initialization + the very first of .media; See https://developers.broadpeak.io/docs/foundations-dash
segments_count = 1 + 1
for s in mpd.periods[0].adaptation_sets[0].representations[0].segment_templates[0].segment_timelines[0].Ss:
segments_count += s.r if s.r else 1
# Populate segment urls.
segment_template = mpd.periods[0].adaptation_sets[0].representations[0].segment_templates[0]
stream_urls: list[str] = []
for index in range(segments_count):
stream_urls.append(segment_template.media.replace('$Number$', str(index)))
elif mime_type == StreamManifestMimeType.JSON.value:
# Stream Manifest is base64 encoded.
manifest_parsed: str = base64.b64decode(manifest).decode("utf-8")
2024-01-13 11:57:11 +01:00
# JSON string to object.
stream_manifest = json.loads(manifest_parsed)
# TODO: Handle more than one dowload URL
2024-01-13 16:26:33 +01:00
stream_urls: str = stream_manifest["urls"]
2024-01-13 11:57:11 +01:00
codecs: str = stream_manifest["codecs"]
mime_type: str = stream_manifest["mimeType"]
encryption_type: str = stream_manifest["encryptionType"]
encryption_key: str | None = (
stream_manifest["encryptionKey"] if self.is_encrypted(encryption_type) else None
)
2024-01-13 16:26:33 +01:00
elif mime_type == StreamManifestMimeType.VIDEO.value:
# Parse M3U8 video playlist
m3u8_variant: m3u8.M3U8 = m3u8.load(manifest)
settings: Settings = Settings()
# Find the desired video resolution or the next best one.
m3u8_playlist, codecs = self._extract_video_stream(m3u8_variant, settings.data.quality_video.value)
# Populate urls.
stream_urls: list[str] = m3u8_playlist.files
# TODO: Handle encryption key. But I have never seen an encrypted file so far.
encryption_type: str = "NONE"
encryption_key: str | None = None
2024-01-13 11:57:11 +01:00
else:
raise UnknownManifestFormat
2024-01-13 16:26:33 +01:00
file_extension: str = self.get_file_extension(stream_urls[0], codecs)
2024-01-13 11:57:11 +01:00
result: StreamManifest = StreamManifest(
2024-01-13 16:26:33 +01:00
stream_urls=stream_urls,
2024-01-13 11:57:11 +01:00
codecs=codecs,
file_extension=file_extension,
encryption_type=encryption_type,
encryption_key=encryption_key,
2024-01-13 16:26:33 +01:00
mime_type=mime_type
2024-01-13 11:57:11 +01:00
)
return result
2024-01-13 16:26:33 +01:00
def _extract_video_stream(self, m3u8_variant: m3u8.M3U8, quality: str) -> (m3u8.M3U8 | bool, str):
m3u8_playlist: m3u8.M3U8 | bool = False
resolution_best: int = 0
mime_type: str = ""
if m3u8_variant.is_variant:
for playlist in m3u8_variant.playlists:
if resolution_best < playlist.stream_info.resolution[1]:
resolution_best = playlist.stream_info.resolution[1]
m3u8_playlist = m3u8.load(playlist.uri)
mime_type = playlist.stream_info.codecs
if quality == playlist.stream_info.resolution[1]:
break
return m3u8_playlist, mime_type