Merge pull request #376 from exislow/358-unable-to-download-album-with-too-many-artists

🛠️ Fixed path length sanitation. Also refactored the path helper to u…
This commit is contained in:
exislow
2025-02-28 16:57:42 +01:00
committed by GitHub
3 changed files with 82 additions and 57 deletions
+1
View File
@@ -11,6 +11,7 @@ BLOCKS: int = 1024
CHUNK_SIZE: int = BLOCK_SIZE * BLOCKS CHUNK_SIZE: int = BLOCK_SIZE * BLOCKS
PLAYLIST_EXTENSION: str = ".m3u" PLAYLIST_EXTENSION: str = ".m3u"
PLAYLIST_PREFIX: str = "_" PLAYLIST_PREFIX: str = "_"
FILENAME_LENGTH_MAX: int = 255
class QualityVideo(StrEnum): class QualityVideo(StrEnum):
+4 -4
View File
@@ -338,7 +338,7 @@ class Download:
).absolute() ).absolute()
# Sanitize final path_file to fit into OS boundaries. # Sanitize final path_file to fit into OS boundaries.
path_media_dst = pathlib.Path(path_file_sanitize(str(path_media_dst), adapt=True)) path_media_dst = pathlib.Path(path_file_sanitize(path_media_dst, adapt=True))
# Compute if and how downloads need to be skipped. # Compute if and how downloads need to be skipped.
skip_download: bool = False skip_download: bool = False
@@ -352,7 +352,7 @@ class Download:
path_media_track_dir: pathlib.Path = ( path_media_track_dir: pathlib.Path = (
pathlib.Path(self.path_base).expanduser() / (file_name_track_dir_relative + file_extension_dummy) pathlib.Path(self.path_base).expanduser() / (file_name_track_dir_relative + file_extension_dummy)
).absolute() ).absolute()
path_media_track_dir = pathlib.Path(path_file_sanitize(str(path_media_track_dir), adapt=True)) path_media_track_dir = pathlib.Path(path_file_sanitize(path_media_track_dir, adapt=True))
file_exists_track_dir: bool = check_file_exists(path_media_track_dir, extension_ignore=False) file_exists_track_dir: bool = check_file_exists(path_media_track_dir, extension_ignore=False)
file_exists_playlist_dir: bool = ( file_exists_playlist_dir: bool = (
not file_exists_track_dir and skip_file and not path_media_dst.is_symlink() not file_exists_track_dir and skip_file and not path_media_dst.is_symlink()
@@ -404,7 +404,7 @@ class Download:
# Compute file name, sanitize once again and create destination directory # Compute file name, sanitize once again and create destination directory
path_media_dst = path_media_dst.with_suffix(file_extension) path_media_dst = path_media_dst.with_suffix(file_extension)
path_media_dst = pathlib.Path(path_file_sanitize(str(path_media_dst), adapt=True)) path_media_dst = pathlib.Path(path_file_sanitize(path_media_dst, adapt=True))
os.makedirs(path_media_dst.parent, exist_ok=True) os.makedirs(path_media_dst.parent, exist_ok=True)
if not skip_download: if not skip_download:
@@ -493,7 +493,7 @@ class Download:
path_media_dst: pathlib.Path = ( path_media_dst: pathlib.Path = (
pathlib.Path(self.path_base).expanduser() / (file_name_relative + file_extension) pathlib.Path(self.path_base).expanduser() / (file_name_relative + file_extension)
).absolute() ).absolute()
path_media_dst = pathlib.Path(path_file_sanitize(str(path_media_dst), adapt=True)) path_media_dst = pathlib.Path(path_file_sanitize(path_media_dst, adapt=True))
os.makedirs(path_media_dst.parent, exist_ok=True) os.makedirs(path_media_dst.parent, exist_ok=True)
+77 -53
View File
@@ -4,6 +4,10 @@ import pathlib
import posixpath import posixpath
import re import re
import sys import sys
from collections.abc import Generator
from copy import deepcopy
from pathlib import Path
from typing import Any
from urllib.parse import unquote, urlsplit from urllib.parse import unquote, urlsplit
from pathvalidate import sanitize_filename, sanitize_filepath from pathvalidate import sanitize_filename, sanitize_filepath
@@ -12,7 +16,7 @@ from tidalapi import Album, Mix, Playlist, Track, UserPlaylist, Video
from tidalapi.media import AudioExtensions from tidalapi.media import AudioExtensions
from tidal_dl_ng import __name_display__ from tidal_dl_ng import __name_display__
from tidal_dl_ng.constants import FILENAME_SANITIZE_PLACEHOLDER, UNIQUIFY_THRESHOLD, MediaType from tidal_dl_ng.constants import FILENAME_LENGTH_MAX, FILENAME_SANITIZE_PLACEHOLDER, UNIQUIFY_THRESHOLD, MediaType
from tidal_dl_ng.helper.tidal import name_builder_album_artist, name_builder_artist, name_builder_title from tidal_dl_ng.helper.tidal import name_builder_album_artist, name_builder_artist, name_builder_title
@@ -203,82 +207,95 @@ def get_format_template(
return result return result
def path_file_sanitize(path_file: str, adapt: bool = False, uniquify: bool = False) -> (bool, str): def path_file_sanitize(path_file: pathlib.Path, adapt: bool = False, uniquify: bool = True) -> pathlib.Path:
# Split into path and filename sanitized_path_file: pathlib.Path = pathlib.Path(path_file.root)
pathname, filename = os.path.split(path_file) # Get each directory name separately (first value in tuple; second value is for the file suffix).
file_extension: str = pathlib.Path(path_file).suffix to_sanitize: [(str, str)] = []
receding_is_first: bool = True
# Sanitize path for i in receding_path(path_file):
if receding_is_first:
receding_is_first = False
to_sanitize.append((i.stem, i.suffix))
else:
to_sanitize.append((i.name, ""))
to_sanitize.reverse()
for name, suffix in to_sanitize:
# Sanitize names: We need first top make sure that none file / directory name has bad chars or is longer than 255 chars.
try:
# sanitize_filename can shorten the file name actually
filename_sanitized: str = sanitize_filename(
name + suffix, replacement_text=" ", validate_after_sanitize=True, platform="auto"
)
# Check if the file extension was removed by shortening the filename length
if not filename_sanitized.endswith(suffix):
# Add the original file extension
file_suffix: str = FILENAME_SANITIZE_PLACEHOLDER + path_file.suffix
filename_sanitized = filename_sanitized[: -len(file_suffix)] + file_suffix
except ValidationError as e:
if adapt:
# TODO: Implement proper exception handling and logging.
# Hacky stuff, since the sanitizing function does not shorten the filename (filename too long)
if str(e).startswith("[PV1101]"):
byte_ct: int = len(name.encode(sys.getfilesystemencoding())) - FILENAME_LENGTH_MAX
filename_sanitized = (
name[: -byte_ct - len(FILENAME_SANITIZE_PLACEHOLDER) - len(suffix)]
+ FILENAME_SANITIZE_PLACEHOLDER
+ suffix
)
else:
raise
else:
raise
finally:
sanitized_path_file = sanitized_path_file / filename_sanitized
# Sanitize the whole path. The whole path with filename is not allowed to be longer then the max path length depending on the OS.
try: try:
pathname_sanitized: str = sanitize_filepath( sanitized_path_file: str = sanitize_filepath(
pathname, replacement_text=" ", validate_after_sanitize=True, platform="auto" sanitized_path_file, replacement_text=" ", validate_after_sanitize=True, platform="auto"
) )
except ValidationError: except ValidationError as e:
# If adaption of path is allowed in case of an error set path to HOME. # If adaption of path is allowed in case of an error set path to HOME.
if adapt: if adapt:
pathname_sanitized: str = str(pathlib.Path.home()) if str(e).startswith("[PV1101]"):
sanitized_path_file = pathlib.Path.home() / sanitized_path_file.name
else:
raise
else: else:
raise raise
# Sanitize filename
try:
filename_sanitized: str = sanitize_filename(
filename, replacement_text=" ", validate_after_sanitize=True, platform="auto"
)
# Check if the file extension was removed by shortening the filename length
if not filename_sanitized.endswith(file_extension):
# Add the original file extension
file_suffix: str = FILENAME_SANITIZE_PLACEHOLDER + file_extension
filename_sanitized = filename_sanitized[: -len(file_suffix)] + file_suffix
except ValidationError as e:
# TODO: Implement proper exception handling and logging.
# Hacky stuff, since the sanitizing function does not shorten the filename somehow (bug?)
# TODO: Remove after pathvalidate update.
# If filename too long
if e.description.startswith("[PV1101]"):
byte_ct: int = len(filename.encode("utf-8")) - 255
filename_sanitized = (
filename[: -byte_ct - len(FILENAME_SANITIZE_PLACEHOLDER) - len(file_extension)]
+ FILENAME_SANITIZE_PLACEHOLDER
+ file_extension
)
else:
print(e)
# Join path and filename
result: str = os.path.join(pathname_sanitized, filename_sanitized)
# Uniquify # Uniquify
if uniquify: if uniquify:
unique_suffix: str = file_unique_suffix(result) unique_suffix: str = file_unique_suffix(sanitized_path_file)
if unique_suffix: if unique_suffix:
file_suffix = unique_suffix + file_extension file_suffix = unique_suffix + sanitized_path_file.suffix
# For most OS filename has a character limit of 255. # For most OS filename has a character limit of 255.
filename_sanitized = ( sanitized_path_file = (
filename_sanitized[: -len(file_suffix)] + file_suffix sanitized_path_file.parent / (str(sanitized_path_file.stem)[: -len(file_suffix)] + file_suffix)
if len(filename_sanitized + unique_suffix) > 255 if len(str(sanitized_path_file.parent / (sanitized_path_file.stem + unique_suffix)))
else filename_sanitized[: -len(file_extension)] + file_suffix > FILENAME_LENGTH_MAX
else sanitized_path_file.parent / (sanitized_path_file.stem + unique_suffix)
) )
# Join path and filename return sanitized_path_file
result = os.path.join(pathname_sanitized, filename_sanitized)
return result
def file_unique_suffix(path_file: str, seperator: str = "_") -> str: def file_unique_suffix(path_file: pathlib.Path, seperator: str = "_") -> str:
threshold_zfill: int = len(str(UNIQUIFY_THRESHOLD)) threshold_zfill: int = len(str(UNIQUIFY_THRESHOLD))
count: int = 0 count: int = 0
path_file_tmp: str = path_file path_file_tmp: pathlib.Path = deepcopy(path_file)
unique_suffix: str = "" unique_suffix: str = ""
while check_file_exists(path_file_tmp) and count < UNIQUIFY_THRESHOLD: while check_file_exists(path_file_tmp) and count < UNIQUIFY_THRESHOLD:
count += 1 count += 1
unique_suffix = seperator + str(count).zfill(threshold_zfill) unique_suffix = seperator + str(count).zfill(threshold_zfill)
filename, file_extension = os.path.splitext(path_file_tmp) path_file_tmp = path_file.parent / (path_file.stem + unique_suffix + path_file.suffix)
path_file_tmp = filename + unique_suffix + file_extension
return unique_suffix return unique_suffix
@@ -325,3 +342,10 @@ def url_to_filename(url: str) -> str:
raise ValueError # reject '%2f' or 'dir%5Cbasename.ext' on Windows raise ValueError # reject '%2f' or 'dir%5Cbasename.ext' on Windows
return basename return basename
def receding_path(p: pathlib.Path) -> Generator[Path | Any, Any, None]:
while str(p) != p.root:
yield p
p = p.parent