🛠️ Fixed path length sanitation. Also refactored the path helper to use pathlib.Path instead of str. Fixed #358
This commit is contained in:
@@ -11,6 +11,7 @@ BLOCKS: int = 1024
|
|||||||
CHUNK_SIZE: int = BLOCK_SIZE * BLOCKS
|
CHUNK_SIZE: int = BLOCK_SIZE * BLOCKS
|
||||||
PLAYLIST_EXTENSION: str = ".m3u"
|
PLAYLIST_EXTENSION: str = ".m3u"
|
||||||
PLAYLIST_PREFIX: str = "_"
|
PLAYLIST_PREFIX: str = "_"
|
||||||
|
FILENAME_LENGTH_MAX: int = 255
|
||||||
|
|
||||||
|
|
||||||
class QualityVideo(StrEnum):
|
class QualityVideo(StrEnum):
|
||||||
|
|||||||
@@ -338,7 +338,7 @@ class Download:
|
|||||||
).absolute()
|
).absolute()
|
||||||
|
|
||||||
# Sanitize final path_file to fit into OS boundaries.
|
# Sanitize final path_file to fit into OS boundaries.
|
||||||
path_media_dst = pathlib.Path(path_file_sanitize(str(path_media_dst), adapt=True))
|
path_media_dst = pathlib.Path(path_file_sanitize(path_media_dst, adapt=True))
|
||||||
|
|
||||||
# Compute if and how downloads need to be skipped.
|
# Compute if and how downloads need to be skipped.
|
||||||
skip_download: bool = False
|
skip_download: bool = False
|
||||||
@@ -352,7 +352,7 @@ class Download:
|
|||||||
path_media_track_dir: pathlib.Path = (
|
path_media_track_dir: pathlib.Path = (
|
||||||
pathlib.Path(self.path_base).expanduser() / (file_name_track_dir_relative + file_extension_dummy)
|
pathlib.Path(self.path_base).expanduser() / (file_name_track_dir_relative + file_extension_dummy)
|
||||||
).absolute()
|
).absolute()
|
||||||
path_media_track_dir = pathlib.Path(path_file_sanitize(str(path_media_track_dir), adapt=True))
|
path_media_track_dir = pathlib.Path(path_file_sanitize(path_media_track_dir, adapt=True))
|
||||||
file_exists_track_dir: bool = check_file_exists(path_media_track_dir, extension_ignore=False)
|
file_exists_track_dir: bool = check_file_exists(path_media_track_dir, extension_ignore=False)
|
||||||
file_exists_playlist_dir: bool = (
|
file_exists_playlist_dir: bool = (
|
||||||
not file_exists_track_dir and skip_file and not path_media_dst.is_symlink()
|
not file_exists_track_dir and skip_file and not path_media_dst.is_symlink()
|
||||||
@@ -404,7 +404,7 @@ class Download:
|
|||||||
|
|
||||||
# Compute file name, sanitize once again and create destination directory
|
# Compute file name, sanitize once again and create destination directory
|
||||||
path_media_dst = path_media_dst.with_suffix(file_extension)
|
path_media_dst = path_media_dst.with_suffix(file_extension)
|
||||||
path_media_dst = pathlib.Path(path_file_sanitize(str(path_media_dst), adapt=True))
|
path_media_dst = pathlib.Path(path_file_sanitize(path_media_dst, adapt=True))
|
||||||
os.makedirs(path_media_dst.parent, exist_ok=True)
|
os.makedirs(path_media_dst.parent, exist_ok=True)
|
||||||
|
|
||||||
if not skip_download:
|
if not skip_download:
|
||||||
@@ -493,7 +493,7 @@ class Download:
|
|||||||
path_media_dst: pathlib.Path = (
|
path_media_dst: pathlib.Path = (
|
||||||
pathlib.Path(self.path_base).expanduser() / (file_name_relative + file_extension)
|
pathlib.Path(self.path_base).expanduser() / (file_name_relative + file_extension)
|
||||||
).absolute()
|
).absolute()
|
||||||
path_media_dst = pathlib.Path(path_file_sanitize(str(path_media_dst), adapt=True))
|
path_media_dst = pathlib.Path(path_file_sanitize(path_media_dst, adapt=True))
|
||||||
|
|
||||||
os.makedirs(path_media_dst.parent, exist_ok=True)
|
os.makedirs(path_media_dst.parent, exist_ok=True)
|
||||||
|
|
||||||
|
|||||||
+67
-43
@@ -4,6 +4,10 @@ import pathlib
|
|||||||
import posixpath
|
import posixpath
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
|
from collections.abc import Generator
|
||||||
|
from copy import deepcopy
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
from urllib.parse import unquote, urlsplit
|
from urllib.parse import unquote, urlsplit
|
||||||
|
|
||||||
from pathvalidate import sanitize_filename, sanitize_filepath
|
from pathvalidate import sanitize_filename, sanitize_filepath
|
||||||
@@ -12,7 +16,7 @@ from tidalapi import Album, Mix, Playlist, Track, UserPlaylist, Video
|
|||||||
from tidalapi.media import AudioExtensions
|
from tidalapi.media import AudioExtensions
|
||||||
|
|
||||||
from tidal_dl_ng import __name_display__
|
from tidal_dl_ng import __name_display__
|
||||||
from tidal_dl_ng.constants import FILENAME_SANITIZE_PLACEHOLDER, UNIQUIFY_THRESHOLD, MediaType
|
from tidal_dl_ng.constants import FILENAME_LENGTH_MAX, FILENAME_SANITIZE_PLACEHOLDER, UNIQUIFY_THRESHOLD, MediaType
|
||||||
from tidal_dl_ng.helper.tidal import name_builder_album_artist, name_builder_artist, name_builder_title
|
from tidal_dl_ng.helper.tidal import name_builder_album_artist, name_builder_artist, name_builder_title
|
||||||
|
|
||||||
|
|
||||||
@@ -203,82 +207,95 @@ def get_format_template(
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
def path_file_sanitize(path_file: str, adapt: bool = False, uniquify: bool = False) -> (bool, str):
|
def path_file_sanitize(path_file: pathlib.Path, adapt: bool = False, uniquify: bool = True) -> pathlib.Path:
|
||||||
# Split into path and filename
|
sanitized_path_file: pathlib.Path = pathlib.Path(path_file.root)
|
||||||
pathname, filename = os.path.split(path_file)
|
# Get each directory name separately (first value in tuple; second value is for the file suffix).
|
||||||
file_extension: str = pathlib.Path(path_file).suffix
|
to_sanitize: [(str, str)] = []
|
||||||
|
receding_is_first: bool = True
|
||||||
|
|
||||||
# Sanitize path
|
for i in receding_path(path_file):
|
||||||
try:
|
if receding_is_first:
|
||||||
pathname_sanitized: str = sanitize_filepath(
|
receding_is_first = False
|
||||||
pathname, replacement_text=" ", validate_after_sanitize=True, platform="auto"
|
|
||||||
)
|
to_sanitize.append((i.stem, i.suffix))
|
||||||
except ValidationError:
|
|
||||||
# If adaption of path is allowed in case of an error set path to HOME.
|
|
||||||
if adapt:
|
|
||||||
pathname_sanitized: str = str(pathlib.Path.home())
|
|
||||||
else:
|
else:
|
||||||
raise
|
to_sanitize.append((i.name, ""))
|
||||||
|
|
||||||
# Sanitize filename
|
to_sanitize.reverse()
|
||||||
|
|
||||||
|
for name, suffix in to_sanitize:
|
||||||
|
# Sanitize names: We need first top make sure that none file / directory name has bad chars or is longer than 255 chars.
|
||||||
try:
|
try:
|
||||||
|
# sanitize_filename can shorten the file name actually
|
||||||
filename_sanitized: str = sanitize_filename(
|
filename_sanitized: str = sanitize_filename(
|
||||||
filename, replacement_text=" ", validate_after_sanitize=True, platform="auto"
|
name + suffix, replacement_text=" ", validate_after_sanitize=True, platform="auto"
|
||||||
)
|
)
|
||||||
|
|
||||||
# Check if the file extension was removed by shortening the filename length
|
# Check if the file extension was removed by shortening the filename length
|
||||||
if not filename_sanitized.endswith(file_extension):
|
if not filename_sanitized.endswith(suffix):
|
||||||
# Add the original file extension
|
# Add the original file extension
|
||||||
file_suffix: str = FILENAME_SANITIZE_PLACEHOLDER + file_extension
|
file_suffix: str = FILENAME_SANITIZE_PLACEHOLDER + path_file.suffix
|
||||||
filename_sanitized = filename_sanitized[: -len(file_suffix)] + file_suffix
|
filename_sanitized = filename_sanitized[: -len(file_suffix)] + file_suffix
|
||||||
except ValidationError as e:
|
except ValidationError as e:
|
||||||
|
if adapt:
|
||||||
# TODO: Implement proper exception handling and logging.
|
# TODO: Implement proper exception handling and logging.
|
||||||
# Hacky stuff, since the sanitizing function does not shorten the filename somehow (bug?)
|
# Hacky stuff, since the sanitizing function does not shorten the filename (filename too long)
|
||||||
# TODO: Remove after pathvalidate update.
|
if str(e).startswith("[PV1101]"):
|
||||||
# If filename too long
|
byte_ct: int = len(name.encode(sys.getfilesystemencoding())) - FILENAME_LENGTH_MAX
|
||||||
if e.description.startswith("[PV1101]"):
|
|
||||||
byte_ct: int = len(filename.encode("utf-8")) - 255
|
|
||||||
filename_sanitized = (
|
filename_sanitized = (
|
||||||
filename[: -byte_ct - len(FILENAME_SANITIZE_PLACEHOLDER) - len(file_extension)]
|
name[: -byte_ct - len(FILENAME_SANITIZE_PLACEHOLDER) - len(suffix)]
|
||||||
+ FILENAME_SANITIZE_PLACEHOLDER
|
+ FILENAME_SANITIZE_PLACEHOLDER
|
||||||
+ file_extension
|
+ suffix
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
print(e)
|
raise
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
sanitized_path_file = sanitized_path_file / filename_sanitized
|
||||||
|
|
||||||
# Join path and filename
|
# Sanitize the whole path. The whole path with filename is not allowed to be longer then the max path length depending on the OS.
|
||||||
result: str = os.path.join(pathname_sanitized, filename_sanitized)
|
try:
|
||||||
|
sanitized_path_file: str = sanitize_filepath(
|
||||||
|
sanitized_path_file, replacement_text=" ", validate_after_sanitize=True, platform="auto"
|
||||||
|
)
|
||||||
|
except ValidationError as e:
|
||||||
|
# If adaption of path is allowed in case of an error set path to HOME.
|
||||||
|
if adapt:
|
||||||
|
if str(e).startswith("[PV1101]"):
|
||||||
|
sanitized_path_file = pathlib.Path.home() / sanitized_path_file.name
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
|
||||||
# Uniquify
|
# Uniquify
|
||||||
if uniquify:
|
if uniquify:
|
||||||
unique_suffix: str = file_unique_suffix(result)
|
unique_suffix: str = file_unique_suffix(sanitized_path_file)
|
||||||
|
|
||||||
if unique_suffix:
|
if unique_suffix:
|
||||||
file_suffix = unique_suffix + file_extension
|
file_suffix = unique_suffix + sanitized_path_file.suffix
|
||||||
# For most OS filename has a character limit of 255.
|
# For most OS filename has a character limit of 255.
|
||||||
filename_sanitized = (
|
sanitized_path_file = (
|
||||||
filename_sanitized[: -len(file_suffix)] + file_suffix
|
sanitized_path_file.parent / (str(sanitized_path_file.stem)[: -len(file_suffix)] + file_suffix)
|
||||||
if len(filename_sanitized + unique_suffix) > 255
|
if len(str(sanitized_path_file.parent / (sanitized_path_file.stem + unique_suffix)))
|
||||||
else filename_sanitized[: -len(file_extension)] + file_suffix
|
> FILENAME_LENGTH_MAX
|
||||||
|
else sanitized_path_file.parent / (sanitized_path_file.stem + unique_suffix)
|
||||||
)
|
)
|
||||||
|
|
||||||
# Join path and filename
|
return sanitized_path_file
|
||||||
result = os.path.join(pathname_sanitized, filename_sanitized)
|
|
||||||
|
|
||||||
return result
|
|
||||||
|
|
||||||
|
|
||||||
def file_unique_suffix(path_file: str, seperator: str = "_") -> str:
|
def file_unique_suffix(path_file: pathlib.Path, seperator: str = "_") -> str:
|
||||||
threshold_zfill: int = len(str(UNIQUIFY_THRESHOLD))
|
threshold_zfill: int = len(str(UNIQUIFY_THRESHOLD))
|
||||||
count: int = 0
|
count: int = 0
|
||||||
path_file_tmp: str = path_file
|
path_file_tmp: pathlib.Path = deepcopy(path_file)
|
||||||
unique_suffix: str = ""
|
unique_suffix: str = ""
|
||||||
|
|
||||||
while check_file_exists(path_file_tmp) and count < UNIQUIFY_THRESHOLD:
|
while check_file_exists(path_file_tmp) and count < UNIQUIFY_THRESHOLD:
|
||||||
count += 1
|
count += 1
|
||||||
unique_suffix = seperator + str(count).zfill(threshold_zfill)
|
unique_suffix = seperator + str(count).zfill(threshold_zfill)
|
||||||
filename, file_extension = os.path.splitext(path_file_tmp)
|
path_file_tmp = path_file.parent / (path_file.stem + unique_suffix + path_file.suffix)
|
||||||
path_file_tmp = filename + unique_suffix + file_extension
|
|
||||||
|
|
||||||
return unique_suffix
|
return unique_suffix
|
||||||
|
|
||||||
@@ -325,3 +342,10 @@ def url_to_filename(url: str) -> str:
|
|||||||
raise ValueError # reject '%2f' or 'dir%5Cbasename.ext' on Windows
|
raise ValueError # reject '%2f' or 'dir%5Cbasename.ext' on Windows
|
||||||
|
|
||||||
return basename
|
return basename
|
||||||
|
|
||||||
|
|
||||||
|
def receding_path(p: pathlib.Path) -> Generator[Path | Any, Any, None]:
|
||||||
|
while str(p) != p.root:
|
||||||
|
yield p
|
||||||
|
|
||||||
|
p = p.parent
|
||||||
|
|||||||
Reference in New Issue
Block a user