diff --git a/votify/downloader/__init__.py b/votify/downloader/__init__.py new file mode 100644 index 0000000..d39b625 --- /dev/null +++ b/votify/downloader/__init__.py @@ -0,0 +1,6 @@ +from .audio import * +from .base import * +from .downloader import * +from .enums import * +from .types import * +from .video import * diff --git a/votify/downloader/audio.py b/votify/downloader/audio.py new file mode 100644 index 0000000..793303a --- /dev/null +++ b/votify/downloader/audio.py @@ -0,0 +1,271 @@ +import asyncio +import logging +from pathlib import Path + +from Crypto.Cipher import AES +from yt_dlp import YoutubeDL +from yt_dlp.downloader.http import HttpFD + +from ..interface.types import SpotifyMedia +from .base import SpotifyBaseDownloader +from .enums import AudioDownloadMode, AudioRemuxMode +from .types import DownloadItem + +logger = logging.getLogger(__name__) + + +class SpotifyAudioDownloader(SpotifyBaseDownloader): + def __init__( + self, + base: SpotifyBaseDownloader, + download_mode: AudioDownloadMode = AudioDownloadMode.YTDLP, + remux_mode: AudioRemuxMode = AudioRemuxMode.FFMPEG, + ) -> None: + self.__dict__.update(base.__dict__) + + self.download_mode = download_mode + self.remux_mode = remux_mode + + async def download_stream( + self, + output_path: str, + stream_url: str, + ) -> None: + logger.info(f"Downloading audio stream from '{stream_url}' to '{output_path}'") + + if self.download_mode == AudioDownloadMode.YTDLP: + await asyncio.to_thread(self._download_with_ytdlp, stream_url, output_path) + elif self.download_mode == AudioDownloadMode.ARIA2C: + await self._download_with_aria2c(stream_url, output_path) + else: + await self._download_with_curl(stream_url, output_path) + + def _download_with_ytdlp(self, stream_url: str, output_path: str) -> None: + Path(output_path).parent.mkdir(parents=True, exist_ok=True) + with YoutubeDL( + { + "quiet": True, + "no_warnings": True, + "noprogress": self.silent, + } + ) as ydl: + http_downloader = HttpFD(ydl, ydl.params) + http_downloader.download( + output_path, + { + "url": stream_url, + }, + ) + + async def _download_with_aria2c(self, stream_url: str, output_path: str) -> None: + output_path_obj = Path(output_path) + output_path_obj.parent.mkdir(parents=True, exist_ok=True) + + await self.run_async_command( + self.aria2c_full_path, + "--no-conf", + "--download-result=hide", + "--console-log-level=error", + "--summary-interval=0", + "--file-allocation=none", + stream_url, + "--dir", + output_path_obj.parent, + "--out", + output_path_obj.name, + silent=self.silent, + ) + + print("\r", end="") + + async def _download_with_curl(self, stream_url: str, output_path: str) -> None: + Path(output_path).parent.mkdir(parents=True, exist_ok=True) + + await self.run_async_command( + self.curl_full_path, + "-sSL", + "-o", + output_path, + stream_url, + silent=self.silent, + ) + + print("\r", end="") + + async def stage( + self, + encrypted_path: str, + decrypted_path: str, + staged_path: str, + decryption_key: bytes | str, + ) -> None: + logger.debug(f"Staging audio: {staged_path}") + decryption_key_hex = ( + decryption_key.hex() + if isinstance(decryption_key, bytes) + else decryption_key + ) + + if staged_path.lower().endswith(".ogg"): + await asyncio.to_thread( + self._decrypt_playplay, + decryption_key, + encrypted_path, + staged_path, + ) + elif self.remux_mode == AudioRemuxMode.FFMPEG: + await self._ffmpeg_remux( + encrypted_path, + staged_path, + decryption_key_hex, + ) + else: + await self._decrypt_mp4decrypt( + encrypted_path, + decrypted_path, + decryption_key_hex, + ) + if self.remux_mode == AudioRemuxMode.MP4BOX: + await self._mp4box_remux( + decrypted_path, + staged_path, + ) + + def _decrypt_playplay( + self, + decryption_key: bytes, + input_path: str, + output_path: str, + ): + cipher = AES.new( + decryption_key, + AES.MODE_CTR, + nonce=bytes.fromhex("72e067fbddcbcf77"), + initial_value=bytes.fromhex("ebe8bc643f630d93"), + ) + + with open(input_path, "rb") as encrypted_file: + encrypted_data = encrypted_file.read() + + with open(output_path, "wb") as decrypted_file: + decrypted_data = cipher.decrypt(encrypted_data) + + offset = decrypted_data.find(b"OggS") + if offset == -1: + msg = "Unable to find ogg header" + raise ValueError(msg) + + decrypted_file.write(decrypted_data[offset:]) + + async def _decrypt_mp4decrypt( + self, + input_path: str, + output_path: str, + decryption_key: str, + ) -> None: + await self.run_async_command( + self.mp4decrypt_full_path, + "--key", + f"1:{decryption_key}", + input_path, + output_path, + silent=self.silent, + ) + + async def _ffmpeg_remux( + self, + input_path: str, + output_path: str, + decryption_key: str, + ) -> None: + await self.run_async_command( + self.ffmpeg_full_path, + "-loglevel", + "error", + "-hide_banner", + "-y", + "-decryption_key", + decryption_key, + "-i", + input_path, + "-c", + "copy", + output_path, + silent=self.silent, + ) + + async def _mp4box_remux( + self, + input_path: str, + output_path: str, + ) -> None: + await self.run_async_command( + self.mp4box_full_path, + "-quiet", + "-itags", + "artist=placeholder", + "-keep-utc", + "-add", + input_path, + "-new", + output_path, + silent=self.silent, + ) + + async def download(self, item: DownloadItem) -> None: + encrypted_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + "encrypted", + "." + item.media.stream_info.audio_track.file_format, + ) + decrypted_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + "decrypted", + "." + item.media.stream_info.audio_track.file_format, + ) + staged_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + "staged", + ".m4a", + ) + await self.download_stream( + encrypted_path, + item.media.stream_info.audio_track.stream_url, + ) + await self.stage( + encrypted_path, + decrypted_path, + staged_path, + item.media.decryption_key.decryption_key, + ) + await self.apply_tags( + staged_path, + item.media.tags, + item.media.cover_url, + ) + + def parse_item(self, media: SpotifyMedia) -> DownloadItem: + item = DownloadItem(media=media) + + item.staged_path = self.get_temp_path( + media.media_id, + item.uuid_, + "staged", + "." + media.stream_info.audio_track.file_format, + ) + item.final_path = self.get_final_path( + media.tags, + "." + media.stream_info.audio_track.file_format, + media.playlist_tags, + ) + if media.playlist_tags: + item.playlist_file_path = self.get_playlist_file_path(media.playlist_tags) + item.synced_lyrics_path = str(Path(item.final_path).with_suffix(".lrc")) + item.cover_path = str(Path(item.final_path).parent / "Cover.jpg") + + logger.debug(f"Parsed audio item: {item}") + + return item diff --git a/votify/downloader/base.py b/votify/downloader/base.py new file mode 100644 index 0000000..d7e4ca6 --- /dev/null +++ b/votify/downloader/base.py @@ -0,0 +1,386 @@ +import asyncio +import base64 +import logging +import re +import shutil +import subprocess +from io import BytesIO +from pathlib import Path + +import httpx +from async_lru import alru_cache +from mutagen.flac import Picture +from mutagen.mp4 import MP4, MP4Cover +from mutagen.oggvorbis import OggVorbis, OggVorbisHeaderError +from PIL import Image + +from ..interface.interface import SpotifyInterface +from ..interface.types import MediaTags, PlaylistTags +from ..utils import CustomStringFormatter +from .constants import ILLEGAL_CHAR_REPLACEMENT, ILLEGAL_CHARS_RE, TEMP_PATH_TEMPLATE + +logger = logging.getLogger(__name__) + + +class SpotifyBaseDownloader: + def __init__( + self, + interface: SpotifyInterface, + output_path: str = "./Spotify", + temp_path: str = ".", + aria2c_path: str = "aria2c", + curl_path: str = "curl", + ffmpeg_path: str = "ffmpeg", + mp4box_path: str = "mp4box", + mp4decrypt_path: str = "mp4decrypt", + shaka_packager_path: str = "packager", + album_folder_template: str = "{album_artist}/{album}", + compilation_folder_template: str = "Compilations/{album}", + no_album_folder_template: str = "{artist}/Unknown Album", + single_disc_file_template: str = "{track:02d} {title}", + multi_disc_file_template: str = "{disc}-{track:02d} {title}", + no_album_file_template: str = "{title}", + playlist_file_template: str = "Playlists/{playlist_artist}/{playlist_title}", + date_tag_template: str = "%Y-%m-%dT%H:%M:%SZ", + exclude_tags: list[str] | None = None, + truncate: int | None = None, + silent: bool = False, + skip_cleanup: bool = False, + ) -> None: + self.interface = interface + self.output_path = output_path + self.temp_path = temp_path + self.aria2c_path = aria2c_path + self.curl_path = curl_path + self.ffmpeg_path = ffmpeg_path + self.mp4box_path = mp4box_path + self.mp4decrypt_path = mp4decrypt_path + self.shaka_packager_path = shaka_packager_path + self.album_folder_template = album_folder_template + self.compilation_folder_template = compilation_folder_template + self.no_album_folder_template = no_album_folder_template + self.single_disc_file_template = single_disc_file_template + self.multi_disc_file_template = multi_disc_file_template + self.no_album_file_template = no_album_file_template + self.playlist_file_template = playlist_file_template + self.date_tag_template = date_tag_template + self.exclude_tags = exclude_tags + self.truncate = truncate + self.silent = silent + self.skip_cleanup = skip_cleanup + + self._initialize() + + def _initialize(self) -> None: + self._initialize_truncate() + self._initialize_full_binaries_path() + + def _initialize_truncate(self) -> None: + if isinstance(self.truncate, int): + self.truncate = None if self.truncate < 4 else self.truncate + + def _initialize_full_binaries_path(self) -> None: + self.aria2c_full_path = shutil.which(self.aria2c_path) + self.curl_full_path = shutil.which(self.curl_path) + self.ffmpeg_full_path = shutil.which(self.ffmpeg_path) + self.mp4box_full_path = shutil.which(self.mp4box_path) + self.mp4decrypt_full_path = shutil.which(self.mp4decrypt_path) + self.shaka_packager_full_path = shutil.which(self.shaka_packager_path) + + def sanitize_string( + self, + dirty_string: str, + file_ext: str = None, + ) -> str: + sanitized_string = re.sub( + ILLEGAL_CHARS_RE, + ILLEGAL_CHAR_REPLACEMENT, + dirty_string, + ) + + if file_ext is None: + sanitized_string = sanitized_string[: self.truncate] + if sanitized_string.endswith("."): + sanitized_string = sanitized_string[:-1] + ILLEGAL_CHAR_REPLACEMENT + else: + if self.truncate is not None: + sanitized_string = sanitized_string[: self.truncate - len(file_ext)] + sanitized_string += file_ext + + return sanitized_string.strip() + + def get_final_path( + self, + tags: MediaTags, + file_extension: str, + playlist_tags: PlaylistTags | None, + ) -> str: + if tags.album: + template_folder_parts = ( + self.compilation_folder_template.split("/") + if tags.compilation + else self.album_folder_template.split("/") + ) + else: + template_folder_parts = self.no_album_folder_template.split("/") + + if tags.album: + template_file_parts = ( + self.multi_disc_file_template.split("/") + if isinstance(tags.disc_total, int) and tags.disc_total > 1 + else self.single_disc_file_template.split("/") + ) + else: + template_file_parts = self.no_album_file_template.split("/") + + template_parts = template_folder_parts + template_file_parts + formatted_parts = [] + + for i, part in enumerate(template_parts): + is_folder = i < len(template_parts) - 1 + formatted_part = CustomStringFormatter().format( + part, + album=(tags.album, "Unknown Album"), + album_artist=(tags.album_artist, "Unknown Artist"), + artist=(tags.artist, "Unknown Artist"), + composer=(tags.composer, "Unknown Composer"), + date=(tags.date, "Unknown Date"), + disc=(tags.disc, ""), + disc_total=(tags.disc_total, ""), + isrc=(tags.isrc, "Unknown ISRC"), + label=(tags.label, "Unknown Label"), + media_id=(tags.media_id, "Unknown Media ID"), + media_type=(tags.media_type, "Unknown Media Type"), + playlist_artist=( + (playlist_tags.artist if playlist_tags else None), + "Unknown Playlist Artist", + ), + playlist_id=( + (playlist_tags.id if playlist_tags else None), + "Unknown Playlist ID", + ), + playlist_title=( + (playlist_tags.title if playlist_tags else None), + "Unknown Playlist Title", + ), + playlist_track=( + (playlist_tags.track if playlist_tags else None), + "", + ), + producer=(tags.producer, "Unknown Producer"), + publisher=(tags.publisher, "Unknown Publisher"), + rating=(tags.rating, "Unknown Rating"), + title=(tags.title, "Unknown Title"), + track=(tags.track, ""), + track_total=(tags.track_total, ""), + ) + sanitized_formatted_part = self.sanitize_string( + formatted_part, + file_extension if not is_folder else None, + ) + formatted_parts.append(sanitized_formatted_part) + + final_path = str(Path(self.output_path, *formatted_parts)) + + logger.debug(f"Generated final path: {final_path}") + + return final_path + + def get_playlist_file_path( + self, + tags: PlaylistTags, + ) -> str: + template_parts = self.playlist_file_template.split("/") + formatted_parts = [] + + for i, part in enumerate(template_parts): + is_folder = i < len(template_parts) - 1 + formatted_part = CustomStringFormatter().format( + part, + artist=(tags.artist, "Unknown Playlist Artist"), + id=(tags.id, "Unknown Playlist ID"), + title=(tags.title, "Unknown Playlist Title"), + track=(tags.track, ""), + track_total=(tags.track_total, ""), + ) + sanitized_formatted_part = self.sanitize_string( + formatted_part, + ".m3u8" if not is_folder else None, + ) + formatted_parts.append(sanitized_formatted_part) + + playlist_file_path = str(Path(self.output_path, *formatted_parts)) + + logger.debug(f"Generated playlist file path: {playlist_file_path}") + + return playlist_file_path + + def update_playlist_file( + self, + playlist_file_path: str, + final_path: str, + playlist_track: int, + ) -> None: + playlist_file_path_obj = Path(playlist_file_path) + final_path_obj = Path(final_path) + output_dir_obj = Path(self.output_path) + + playlist_file_path_obj.parent.mkdir(parents=True, exist_ok=True) + playlist_file_path_parent_parts_len = len(playlist_file_path_obj.parent.parts) + output_path_parts_len = len(output_dir_obj.parts) + + final_path_relative = Path( + ("../" * (playlist_file_path_parent_parts_len - output_path_parts_len)), + *final_path_obj.parts[output_path_parts_len:], + ) + playlist_file_lines = ( + playlist_file_path_obj.open("r", encoding="utf8").readlines() + if playlist_file_path_obj.exists() + else [] + ) + if len(playlist_file_lines) < playlist_track: + playlist_file_lines.extend( + "\n" for _ in range(playlist_track - len(playlist_file_lines)) + ) + + playlist_file_lines[playlist_track - 1] = final_path_relative.as_posix() + "\n" + with playlist_file_path_obj.open("w", encoding="utf8") as playlist_file: + playlist_file.writelines(playlist_file_lines) + + logger.debug( + f"Updated playlist file '{playlist_file_path}' with track {playlist_track}: {final_path_relative.as_posix()}" + ) + + def get_temp_path( + self, + media_id: str, + folder_tag: str, + file_tag: str, + file_extension: str, + ) -> str: + return str( + Path(self.temp_path) + / TEMP_PATH_TEMPLATE.format(folder_tag) + / (f"{media_id}_{file_tag}" + file_extension) + ) + + @alru_cache() + async def get_cover_bytes(self, cover_url) -> bytes | None: + async with httpx.AsyncClient() as client: + response = await client.get(cover_url) + + if response.status_code == 200: + return response.content + + if response.status_code == 404: + return None + + response.raise_for_status() + + async def apply_tags( + self, + input_path: str, + tags: MediaTags, + cover_url: str, + ) -> None: + exclude_tags = self.exclude_tags or [] + filtered_tags = MediaTags( + **{ + k: v + for k, v in tags.__dict__.items() + if v is not None and k not in exclude_tags + } + ) + + cover_bytes = await self.get_cover_bytes(cover_url) + + logger.debug(f"Applying tags to '{input_path}': {filtered_tags}") + + if input_path.lower().endswith(".ogg"): + self._apply_ogg_tags( + input_path, + filtered_tags, + cover_bytes, + exclude_tags, + ) + else: + self._apply_mp4_tags( + input_path, + filtered_tags, + cover_bytes, + exclude_tags, + ) + + def _apply_ogg_tags( + self, + input_path: str, + tags: MediaTags, + cover_bytes: bytes | None, + exclude_tags: list[str], + ) -> None: + file = OggVorbis(input_path) + file.clear() + skip_tagging = "all" in exclude_tags + + if not skip_tagging: + ogg_tags = tags.as_vorbis_tags(self.date_tag_template) + file.update(ogg_tags) + + if not skip_tagging and "cover" not in exclude_tags and cover_bytes: + picture = Picture() + picture.mime = "image/jpeg" + picture.data = cover_bytes + picture.type = 3 + picture.width, picture.height = Image.open(BytesIO(cover_bytes)).size + ogg_tags["METADATA_BLOCK_PICTURE"] = base64.b64encode( + picture.write() + ).decode("ascii") + + try: + file.save() + except OggVorbisHeaderError: + pass + + def _apply_mp4_tags( + self, + input_path: str, + tags: MediaTags, + cover_bytes: bytes | None, + exclude_tags: list[str], + ) -> None: + mp4 = MP4(input_path) + mp4.clear() + skip_tagging = "all" in exclude_tags + + if not skip_tagging: + mp4_tags = tags.as_mp4_tags(self.date_tag_template) + logger.debug(f"MP4 tags to apply: {mp4_tags}") + mp4.update(mp4_tags) + + if not skip_tagging and "cover" not in exclude_tags and cover_bytes: + mp4["covr"] = [ + MP4Cover( + data=cover_bytes, + imageformat=MP4Cover.FORMAT_JPEG, + ) + ] + + mp4.save() + + @staticmethod + async def run_async_command(*args: str, silent: bool = False) -> None: + if silent: + additional_args = { + "stdout": subprocess.DEVNULL, + "stderr": subprocess.DEVNULL, + } + else: + additional_args = {} + + proc = await asyncio.create_subprocess_exec( + *args, + **additional_args, + ) + await proc.communicate() + if proc.returncode != 0: + raise Exception(f'"{args[0]}" exited with code {proc.returncode}') diff --git a/votify/downloader/constants.py b/votify/downloader/constants.py new file mode 100644 index 0000000..377e04e --- /dev/null +++ b/votify/downloader/constants.py @@ -0,0 +1,3 @@ +TEMP_PATH_TEMPLATE = "votify_temp_{}" +ILLEGAL_CHARS_RE = r'[\\/:*?"<>|;]' +ILLEGAL_CHAR_REPLACEMENT = "_" diff --git a/votify/downloader/downloader.py b/votify/downloader/downloader.py new file mode 100644 index 0000000..f490446 --- /dev/null +++ b/votify/downloader/downloader.py @@ -0,0 +1,215 @@ +import logging +import shutil +from pathlib import Path +from typing import AsyncGenerator + +from ..interface.enums import MediaType +from .audio import SpotifyAudioDownloader +from .base import SpotifyBaseDownloader +from .constants import TEMP_PATH_TEMPLATE +from .enums import AudioDownloadMode, AudioRemuxMode, VideoRemuxMode +from .exceptions import ( + VotifyDependencyNotFound, + VotifyMediaFileExists, + VotifySyncedLyricsOnly, +) +from .types import DownloadItem +from .video import SpotifyVideoDownloader + +logger = logging.getLogger(__name__) + + +class SpotifyDownloader: + def __init__( + self, + base: SpotifyBaseDownloader, + audio: SpotifyAudioDownloader, + video: SpotifyVideoDownloader, + no_synced_lyrics_file: bool = False, + save_playlist_file: bool = False, + save_cover_file: bool = False, + overwrite: bool = False, + synced_lyrics_only: bool = False, + skip_processing: bool = False, + skip_cleanup: bool = False, + ) -> None: + self.base = base + self.audio = audio + self.video = video + self.no_synced_lyrics_file = no_synced_lyrics_file + self.save_playlist_file = save_playlist_file + self.save_cover_file = save_cover_file + self.overwrite = overwrite + self.synced_lyrics_only = synced_lyrics_only + self.skip_processing = skip_processing + self.skip_cleanup = skip_cleanup + + async def get_download_item(self, url: str) -> AsyncGenerator[DownloadItem, None]: + async for media in self.base.interface.get_media_by_url(url): + if media.tags.media_type in { + MediaType.SONG, + MediaType.PODCAST, + }: + yield self.audio.parse_item(media) + elif media.tags.media_type in { + MediaType.MUSIC_VIDEO, + MediaType.PODCAST_VIDEO, + }: + yield self.video.parse_item(media) + + async def download(self, item: DownloadItem) -> None: + try: + await self._initial_processing(item) + await self._download(item) + await self._final_processing(item) + finally: + if not self.skip_cleanup: + self._cleanup_temp(item.uuid_) + + async def _download(self, item: DownloadItem) -> None: + if self.synced_lyrics_only: + raise VotifySyncedLyricsOnly() + + if item.final_path and Path(item.final_path).exists() and not self.overwrite: + raise VotifyMediaFileExists(item.final_path) + + if item.media.tags.media_type in { + MediaType.SONG, + MediaType.PODCAST, + }: + if ( + self.audio.download_mode == AudioDownloadMode.ARIA2C + and not self.base.aria2c_full_path + ): + raise VotifyDependencyNotFound("aria2c") + + if ( + self.audio.download_mode == AudioDownloadMode.CURL + and not self.base.curl_full_path + ): + raise VotifyDependencyNotFound("cURL") + + if ( + self.base.interface.song.audio_quality.mp4 + and self.audio.remux_mode == AudioRemuxMode.FFMPEG + and not self.base.ffmpeg_full_path + ): + raise VotifyDependencyNotFound("ffmpeg") + + if ( + self.base.interface.song.audio_quality.mp4 + and self.audio.remux_mode == AudioRemuxMode.MP4BOX + and not self.base.mp4box_full_path + ): + raise VotifyDependencyNotFound("MP4Box") + + if ( + self.base.interface.song.audio_quality.mp4 + and ( + self.audio.remux_mode == AudioRemuxMode.MP4DECRYPT + or self.audio.remux_mode == AudioRemuxMode.MP4BOX + ) + and not self.base.mp4decrypt_full_path + ): + raise VotifyDependencyNotFound("mp4decrypt") + + await self.audio.download(item) + elif item.media.tags.media_type in { + MediaType.MUSIC_VIDEO, + MediaType.PODCAST_VIDEO, + }: + if ( + self.video.remux_mode == VideoRemuxMode.FFMPEG + and not self.base.ffmpeg_full_path + ): + raise VotifyDependencyNotFound("ffmpeg") + + if ( + self.video.remux_mode == VideoRemuxMode.MP4BOX + and not self.base.mp4box_full_path + ): + raise VotifyDependencyNotFound("MP4Box") + + if item.media.decryption_key: + if ( + item.media.stream_info.video_track.file_format == "mp4" + or item.media.stream_info.audio_track.file_format == "mp4" + ) and not self.base.mp4decrypt_full_path: + raise VotifyDependencyNotFound("mp4decrypt") + + if ( + item.media.stream_info.video_track.file_format == "webm" + or item.media.stream_info.audio_track.file_format == "webm" + ) and not self.base.shaka_packager_full_path: + raise VotifyDependencyNotFound("Shaka Packager") + + await self.video.download(item) + + def _cleanup_temp(self, folder_tag: str) -> None: + temp_path = Path(self.base.temp_path) / TEMP_PATH_TEMPLATE.format(folder_tag) + if temp_path.exists() and temp_path.is_dir(): + shutil.rmtree(temp_path, ignore_errors=True) + + async def _initial_processing(self, item: DownloadItem) -> None: + if self.skip_processing: + return + + if item.cover_path and self.save_cover_file: + cover_bytes = await self.base.get_cover_bytes( + item.media.cover_url, + ) + if cover_bytes and (self.overwrite or not Path(item.cover_path).exists()): + self._write_cover_file( + item.cover_path, + cover_bytes, + ) + + if ( + item.synced_lyrics_path + and not self.no_synced_lyrics_file + and item.media.lyrics + and item.media.lyrics.synced + and (self.overwrite or not Path(item.synced_lyrics_path).exists()) + ): + self._write_synced_lyrics_file( + item.synced_lyrics_path, + item.media.lyrics.synced, + ) + + async def _final_processing( + self, + item: DownloadItem, + ) -> None: + if self.skip_processing: + return + + if item.staged_path and item.final_path and Path(item.staged_path).exists(): + self._move_to_final_path( + item.staged_path, + item.final_path, + ) + if item.playlist_file_path and self.save_playlist_file: + self.base.update_playlist_file( + item.playlist_file_path, + item.media.playlist_tags, + ) + + def _write_cover_file(self, cover_path: str, cover_bytes: bytes) -> None: + logger.debug(f"Writing cover: {cover_path}") + + Path(cover_path).parent.mkdir(parents=True, exist_ok=True) + with open(cover_path, "wb") as f: + f.write(cover_bytes) + + def _write_synced_lyrics_file(self, synced_lyrics_path: str, lyrics: str) -> None: + logger.debug(f"Writing synced lyrics: {synced_lyrics_path}") + + Path(synced_lyrics_path).parent.mkdir(parents=True, exist_ok=True) + with open(synced_lyrics_path, "w", encoding="utf-8") as f: + f.write(lyrics) + + def _move_to_final_path(self, staged_path: str, final_path: str) -> None: + logger.debug(f'Moving "{staged_path}" to "{final_path}"') + + Path(final_path).parent.mkdir(parents=True, exist_ok=True) + shutil.move(staged_path, final_path) diff --git a/votify/downloader/enums.py b/votify/downloader/enums.py new file mode 100644 index 0000000..14e6ff3 --- /dev/null +++ b/votify/downloader/enums.py @@ -0,0 +1,18 @@ +from enum import Enum + + +class AudioDownloadMode(Enum): + YTDLP = "ytdlp" + ARIA2C = "aria2c" + CURL = "curl" + + +class AudioRemuxMode(Enum): + FFMPEG = "ffmpeg" + MP4BOX = "mp4box" + MP4DECRYPT = "mp4decrypt" + + +class VideoRemuxMode(Enum): + FFMPEG = "ffmpeg" + MP4BOX = "mp4box" diff --git a/votify/downloader/exceptions.py b/votify/downloader/exceptions.py new file mode 100644 index 0000000..2b48923 --- /dev/null +++ b/votify/downloader/exceptions.py @@ -0,0 +1,24 @@ +from ..utils import VotiyException + + +class VotifyDownloaderException(VotiyException): + pass + + +class VotifyMediaFileExists(VotifyDownloaderException): + def __init__(self, media_path: str): + super().__init__(f"Media file already exists at path: {media_path}") + + self.media_path = media_path + + +class VotifyDependencyNotFound(VotifyDownloaderException): + def __init__(self, dependency: str): + super().__init__(f"Dependency not found: {dependency}") + + self.dependency = dependency + + +class VotifySyncedLyricsOnly(VotifyDownloaderException): + def __init__(self): + super().__init__("Only downloading synced lyrics is supported") diff --git a/votify/downloader/types.py b/votify/downloader/types.py new file mode 100644 index 0000000..3fcf1fc --- /dev/null +++ b/votify/downloader/types.py @@ -0,0 +1,15 @@ +from dataclasses import dataclass +import uuid + +from ..interface.types import SpotifyMedia + + +@dataclass +class DownloadItem: + media: SpotifyMedia + uuid_: str = uuid.uuid4().hex[:8] + staged_path: str = None + final_path: str = None + playlist_file_path: str = None + synced_lyrics_path: str = None + cover_path: str = None diff --git a/votify/downloader/video.py b/votify/downloader/video.py new file mode 100644 index 0000000..7d91d81 --- /dev/null +++ b/votify/downloader/video.py @@ -0,0 +1,273 @@ +import logging +from pathlib import Path + +from yt_dlp import YoutubeDL +from yt_dlp.downloader.fragment import FragmentFD + +from ..interface.types import SpotifyMedia +from .base import SpotifyBaseDownloader +from .enums import VideoRemuxMode +from .types import DownloadItem + +logger = logging.getLogger(__name__) + + +class SpotifyVideoDownloader(SpotifyBaseDownloader): + def __init__( + self, + base: SpotifyBaseDownloader, + remux_mode: VideoRemuxMode = VideoRemuxMode.FFMPEG, + ) -> None: + self.__dict__.update(base.__dict__) + + self.remux_mode = remux_mode + + def _download_stream( + self, + input_path: str, + segment_urls: list[str], + ): + logger.debug(f"Downloading video stream to '{input_path}'") + + Path(input_path).parent.mkdir(parents=True, exist_ok=True) + segments_dict = [ + {"url": url, "frag_index": idx + 1} for idx, url in enumerate(segment_urls) + ] + ctx = { + "filename": str(input_path), + "total_frags": len(segments_dict), + } + info_dict = {} + with YoutubeDL( + { + "quiet": True, + "no_warnings": True, + "noprogress": self.silent, + }, + ) as ydl: + try: + fragment_downloader = FragmentFD(ydl, ydl.params) + fragment_downloader._prepare_and_start_frag_download( + ctx, + info_dict, + ) + fragment_downloader.download_and_append_fragments( + ctx, + segments_dict, + info_dict, + ) + finally: + fragment_downloader._finish_multiline_status() + + async def stage( + self, + encrypted_video_path: str, + encrypted_audio_path: str, + decrypted_video_path: str, + decrypted_audio_path: str, + staged_path: str, + decryption_key: str | None, + key_id: str | None, + ) -> None: + logger.debug(f"Staging video: {staged_path}") + + if decryption_key: + if encrypted_video_path.lower().endswith(".webm"): + await self._shaka_packager_decrypt( + encrypted_video_path, + decrypted_video_path, + decryption_key, + key_id, + ) + else: + await self._mp4decrypt_decrypt( + encrypted_video_path, + decrypted_video_path, + decryption_key, + ) + + if encrypted_audio_path.lower().endswith(".webm"): + await self._shaka_packager_decrypt( + encrypted_audio_path, + decrypted_audio_path, + decryption_key, + key_id, + ) + else: + await self._mp4decrypt_decrypt( + encrypted_audio_path, + decrypted_audio_path, + decryption_key, + ) + + if self.remux_mode == VideoRemuxMode.FFMPEG: + await self._ffmpeg_remux( + decrypted_video_path, + decrypted_audio_path, + staged_path, + ) + else: + await self._mp4box_remux( + decrypted_video_path, + decrypted_audio_path, + staged_path, + ) + + async def _shaka_packager_decrypt( + self, + input_path: str, + output_path: str, + decryption_key: str, + key_id: str, + ): + await self.run_async_command( + self.shaka_packager_full_path, + "--quiet", + f"stream=0,in={input_path},output={output_path}", + "-enable_raw_key_decryption", + "-keys", + f"key_id={key_id}:key={decryption_key}", + silent=self.silent, + ) + + async def _mp4decrypt_decrypt( + self, + input_path: Path, + output_path: Path, + decryption_key: str, + ): + await self.run_async_command( + self.mp4decrypt_full_path, + "--key", + f"1:{decryption_key}", + input_path, + output_path, + silent=self.silent, + ) + + async def _ffmpeg_remux( + self, + input_path_video: str, + input_path_audio: str, + output_path: str, + ): + await self.run_async_command( + self.ffmpeg_full_path, + "-loglevel", + "error", + "-y", + "-i", + input_path_video, + "-i", + input_path_audio, + "-c", + "copy", + "-map", + "0:v:0", + "-map", + "1:a:0", + output_path, + silent=self.silent, + ) + + async def _mp4box_remux( + self, + input_path_video: str, + input_path_audio: str, + output_path: str, + ): + await self.run_async_command( + self.mp4box_full_path, + "-quiet", + "-itags", + "artist=placeholder", + "-keep-utc", + "-add", + input_path_video, + "-add", + input_path_audio, + "-new", + output_path, + silent=self.silent, + ) + + async def download(self, item: DownloadItem) -> None: + encrypted_video_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + f"{item.uuid_}_video_encrypted", + "." + item.media.stream_info.video_track.file_format, + ) + encrypted_audio_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + f"{item.uuid_}_audio_encrypted", + "." + item.media.stream_info.audio_track.file_format, + ) + decrypted_video_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + f"{item.uuid_}_video_decrypted", + "." + item.media.stream_info.video_track.file_format, + ) + decrypted_audio_path = self.get_temp_path( + item.media.media_id, + item.uuid_, + f"{item.uuid_}_audio_decrypted", + "." + item.media.stream_info.audio_track.file_format, + ) + + decryption_key, key_id = ( + ( + item.media.decryption_key.decryption_key, + item.media.decryption_key.key_id, + ) + if item.media.decryption_key + else (None, None) + ) + + self._download_stream( + encrypted_video_path if decryption_key else decrypted_video_path, + item.media.stream_info.video_track.stream_url, + ) + self._download_stream( + encrypted_audio_path if decryption_key else decrypted_audio_path, + item.media.stream_info.audio_track.stream_url, + ) + + await self.stage( + encrypted_video_path=encrypted_video_path, + encrypted_audio_path=encrypted_audio_path, + decrypted_video_path=decrypted_video_path, + decrypted_audio_path=decrypted_audio_path, + staged_path=item.staged_path, + decryption_key=decryption_key, + key_id=key_id, + ) + await self.apply_tags( + item.staged_path, + item.media.tags, + item.media.cover_url, + ) + + def parse_item(self, media: SpotifyMedia) -> DownloadItem: + item = DownloadItem(media=media) + + item.staged_path = self.get_temp_path( + media.media_id, + item.uuid_, + "staged", + ".mp4", + ) + item.final_path = self.get_final_path( + media.tags, + ".mp4", + media.playlist_tags, + ) + if media.playlist_tags: + item.playlist_file_path = self.get_playlist_file_path(media.playlist_tags) + item.cover_path = str(Path(item.final_path).with_suffix(".jpg")) + + logger.debug(f"Parsed video item: {item}") + + return item