import abc import contextlib import os from abc import ABC from pathlib import Path from typing import Dict from typing import Generic from typing import Iterable from typing import Iterator from typing import List from typing import Optional from typing import Set from typing import Tuple from typing import Type from typing import TypeVar from ytdl_sub.config.preset_options import AddsVariablesMixin from ytdl_sub.config.preset_options import Overrides from ytdl_sub.downloaders.generic.validators import MultiUrlValidator from ytdl_sub.downloaders.generic.validators import UrlThumbnailListValidator from ytdl_sub.downloaders.generic.validators import UrlValidator from ytdl_sub.downloaders.ytdl_options_builder import YTDLOptionsBuilder from ytdl_sub.downloaders.ytdlp import YTDLP from ytdl_sub.entries.entry import Entry from ytdl_sub.entries.entry_parent import EntryParent from ytdl_sub.entries.variables.kwargs import COLLECTION_URL from ytdl_sub.entries.variables.kwargs import COMMENTS from ytdl_sub.entries.variables.kwargs import DOWNLOAD_INDEX from ytdl_sub.entries.variables.kwargs import PLAYLIST_ENTRY from ytdl_sub.entries.variables.kwargs import REQUESTED_SUBTITLES from ytdl_sub.entries.variables.kwargs import SOURCE_ENTRY from ytdl_sub.entries.variables.kwargs import SPONSORBLOCK_CHAPTERS from ytdl_sub.entries.variables.kwargs import UPLOAD_DATE_INDEX from ytdl_sub.plugins.plugin import Plugin from ytdl_sub.plugins.plugin import PluginOptions from ytdl_sub.utils.file_handler import FileHandler from ytdl_sub.utils.logger import Logger from ytdl_sub.utils.thumbnail import ThumbnailTypes from ytdl_sub.utils.thumbnail import convert_download_thumbnail from ytdl_sub.utils.thumbnail import download_and_convert_url_thumbnail from ytdl_sub.validators.strict_dict_validator import StrictDictValidator from ytdl_sub.ytdl_additions.enhanced_download_archive import DownloadArchiver from ytdl_sub.ytdl_additions.enhanced_download_archive import EnhancedDownloadArchive # pylint: disable=too-many-instance-attributes download_logger = Logger.get(name="downloader") class DownloaderValidator(StrictDictValidator, AddsVariablesMixin, ABC): """ Placeholder class to define downloader options """ @property @abc.abstractmethod def collection_validator(self) -> MultiUrlValidator: """ Returns ------- MultiUrlValidator To determine how the entries are downloaded """ def added_source_variables(self) -> List[str]: """ Returns ------- Added source variables on the collection """ return self.collection_validator.added_source_variables() def validate_with_variables( self, source_variables: List[str], override_variables: Dict[str, str] ) -> None: """ Validates any source variables added by the collection """ self.collection_validator.validate_with_variables( source_variables=source_variables, override_variables=override_variables ) DownloaderOptionsT = TypeVar("DownloaderOptionsT", bound=DownloaderValidator) class URLDownloadState: def __init__(self, entries_total: int): self.entries_total = entries_total self.entries_downloaded = 0 class EmptyPluginOptions(PluginOptions): _optional_keys = {"no-op"} class BaseDownloaderPlugin(Plugin[EmptyPluginOptions], ABC): def __init__( self, overrides: Overrides, enhanced_download_archive: EnhancedDownloadArchive, ): super().__init__( # Downloader plugins do not have exposed YAML options, so keep it blank. # Use init instead. plugin_options=EmptyPluginOptions(name=self.__class__.__name__, value={}), overrides=overrides, enhanced_download_archive=enhanced_download_archive, ) class BaseDownloader(DownloadArchiver, Generic[DownloaderOptionsT], ABC): downloader_options_type: Type[DownloaderValidator] = DownloaderValidator def __init__( self, download_options: DownloaderOptionsT, enhanced_download_archive: EnhancedDownloadArchive, download_ytdl_options: YTDLOptionsBuilder, metadata_ytdl_options: YTDLOptionsBuilder, overrides: Overrides, ): super().__init__(enhanced_download_archive=enhanced_download_archive) self.download_options = download_options self.overrides = overrides self._download_ytdl_options_builder = download_ytdl_options self._metadata_ytdl_options_builder = metadata_ytdl_options @abc.abstractmethod def download_metadata(self) -> Iterable[Entry]: """Gathers metadata of all entries to download""" @abc.abstractmethod def download(self, entry: Entry) -> Entry: """The function to perform the download of all media entries""" # pylint: disable=no-self-use def added_plugins(self) -> List[BaseDownloaderPlugin]: """Add these plugins from the Downloader to the subscription""" return [] # pylint: enable=no-self-use class YtDlpThumbnailPlugin(BaseDownloaderPlugin): def __init__( self, overrides: Overrides, enhanced_download_archive: EnhancedDownloadArchive, collection_urls: List[UrlValidator], ): super().__init__( overrides=overrides, enhanced_download_archive=enhanced_download_archive, ) self._thumbnails_downloaded: Set[str] = set() self._collection_url_mapping: Dict[str, UrlValidator] = { self.overrides.apply_formatter(collection_url.url): collection_url for collection_url in collection_urls } def _download_parent_thumbnails( self, thumbnail_list_info: UrlThumbnailListValidator, entry: Entry, parent: EntryParent, ) -> None: """ Downloads and moves channel avatar and banner images to the output directory. """ for thumbnail_info in thumbnail_list_info.list: thumbnail_name = self.overrides.apply_formatter(thumbnail_info.name, entry=entry) thumbnail_id = self.overrides.apply_formatter(thumbnail_info.uid) # If latest entry, always update the thumbnail on each entry if thumbnail_id == ThumbnailTypes.LATEST_ENTRY: # Make sure the entry's thumbnail is converted to jpg convert_download_thumbnail(entry, error_if_not_found=False) # always save in dry-run even if it doesn't exist... if self.is_dry_run or os.path.isfile(entry.get_download_thumbnail_path()): self.save_file( file_name=entry.get_download_thumbnail_name(), output_file_name=thumbnail_name, copy_file=True, ) self._thumbnails_downloaded.add(thumbnail_name) continue # If not latest entry and the thumbnail has already been downloaded, then skip if thumbnail_name in self._thumbnails_downloaded: continue if (thumbnail_url := parent.get_thumbnail_url(thumbnail_id=thumbnail_id)) is None: download_logger.debug("Failed to find thumbnail id '%s'", thumbnail_id) continue if download_and_convert_url_thumbnail( thumbnail_url=thumbnail_url, output_thumbnail_path=str(Path(self.working_directory) / thumbnail_name), ): self.save_file(file_name=thumbnail_name) self._thumbnails_downloaded.add(thumbnail_name) else: download_logger.debug("Failed to download thumbnail id '%s'", thumbnail_id) def _download_url_thumbnails(self, collection_url: UrlValidator, entry: Entry): """ After all media entries have been downloaded, post processed, and moved to the output directory, run this function. This lets the downloader add any extra files directly to the output directory, for things like YT channel image, banner. """ if entry.kwargs_contains(PLAYLIST_ENTRY): self._download_parent_thumbnails( thumbnail_list_info=collection_url.playlist_thumbnails, entry=entry, parent=EntryParent( entry.kwargs(PLAYLIST_ENTRY), working_directory=self.working_directory ), ) if entry.kwargs_contains(SOURCE_ENTRY): self._download_parent_thumbnails( thumbnail_list_info=collection_url.source_thumbnails, entry=entry, parent=EntryParent( entry.kwargs(SOURCE_ENTRY), working_directory=self.working_directory ), ) def modify_entry(self, entry: Entry) -> Optional[Entry]: """ Use the entry to download thumbnails (or move if LATEST_ENTRY) """ if entry.kwargs(COLLECTION_URL) in self._collection_url_mapping: self._download_url_thumbnails( collection_url=self._collection_url_mapping[entry.kwargs(COLLECTION_URL)], entry=entry, ) return entry class YtDlpCollectionVariablePlugin(BaseDownloaderPlugin): def __init__( self, overrides: Overrides, enhanced_download_archive: EnhancedDownloadArchive, collection_urls: List[UrlValidator], ): super().__init__( overrides=overrides, enhanced_download_archive=enhanced_download_archive, ) self._thumbnails_downloaded: Set[str] = set() self._collection_url_mapping: Dict[str, UrlValidator] = { self.overrides.apply_formatter(collection_url.url): collection_url for collection_url in collection_urls } def modify_entry_metadata(self, entry: Entry) -> Optional[Entry]: """ Add collection variables to the entry """ collection_url: Optional[UrlValidator] = self._collection_url_mapping.get( entry.kwargs(COLLECTION_URL) ) if collection_url: entry.add_variables(variables_to_add=collection_url.variables.dict_with_format_strings) return entry class YtDlpDownloader(BaseDownloader[DownloaderOptionsT], ABC): """ Class that interacts with ytdl to perform the download of metadata and content, and should translate that to list of Entry objects. """ def added_plugins(self) -> List[Plugin]: """ Adds 1. URL thumbnail download plugin 2. Collection variable plugin to add to each entry """ return [ YtDlpThumbnailPlugin( overrides=self.overrides, enhanced_download_archive=self._enhanced_download_archive, collection_urls=self.collection.urls.list, ), YtDlpCollectionVariablePlugin( overrides=self.overrides, enhanced_download_archive=self._enhanced_download_archive, collection_urls=self.collection.urls.list, ), ] @classmethod def ytdl_option_defaults(cls) -> Dict: """ .. code-block:: yaml ytdl_options: ignoreerrors: True # ignore errors like hidden videos, age restriction, etc """ return {"ignoreerrors": True} def __init__( self, download_options: DownloaderOptionsT, enhanced_download_archive: EnhancedDownloadArchive, download_ytdl_options: YTDLOptionsBuilder, metadata_ytdl_options: YTDLOptionsBuilder, overrides: Overrides, ): """ Parameters ---------- download_options Options validator for this downloader enhanced_download_archive Download archive download_ytdl_options YTDL options builder for downloading media metadata_ytdl_options YTDL options builder for downloading metadata overrides Override variables """ super().__init__( download_options=download_options, enhanced_download_archive=enhanced_download_archive, download_ytdl_options=download_ytdl_options, metadata_ytdl_options=metadata_ytdl_options, overrides=overrides, ) self._downloaded_entries: Set[str] = set() self._url_state: Optional[URLDownloadState] = None @property def download_ytdl_options(self) -> Dict: """ Returns ------- YTLD options dict for downloading """ return ( self._download_ytdl_options_builder.clone() .add(self.ytdl_option_defaults(), before=True) .to_dict() ) @property def metadata_ytdl_options(self) -> Dict: """ Returns ------- YTDL options dict for fetching metadata """ return ( self._metadata_ytdl_options_builder.clone() .add(self.ytdl_option_defaults(), before=True) .to_dict() ) @property def is_dry_run(self) -> bool: """ Returns ------- True if dry-run is enabled. False otherwise. """ return self.download_ytdl_options.get("skip_download", False) @property def is_entry_thumbnails_enabled(self) -> bool: """ Returns ------- True if entry thumbnails should be downloaded. False otherwise. """ return self.download_ytdl_options.get("writethumbnail", False) ############################################################################################### # DOWNLOAD FUNCTIONS def _is_downloaded(self, entry: Entry) -> bool: return entry.ytdl_uid() in self._downloaded_entries def _mark_downloaded(self, entry: Entry) -> None: self._downloaded_entries.add(entry.ytdl_uid()) @property def collection(self) -> MultiUrlValidator: """Return the download options collection""" return self.download_options.collection_validator @contextlib.contextmanager def _separate_download_archives(self, clear_info_json_files: bool = False): """ Separate download archive writing between collection urls. This is so break_on_existing does not break when downloading from subset urls. Parameters ---------- clear_info_json_files Whether to delete info.json files after yield """ archive_path = self.download_ytdl_options.get("download_archive", "") backup_archive_path = f"{archive_path}.backup" # If archive path exists, maintain download archive is enable if archive_file_exists := archive_path and os.path.isfile(archive_path): archive_file_exists = True # If a backup exists, it's the one prior to any downloading, use that. if os.path.isfile(backup_archive_path): FileHandler.copy(src_file_path=backup_archive_path, dst_file_path=archive_path) # If not, create the backup else: FileHandler.copy(src_file_path=archive_path, dst_file_path=backup_archive_path) yield # If an archive path did not exist at first, but now exists, delete it if not archive_file_exists and os.path.isfile(archive_path): FileHandler.delete(file_path=archive_path) # If the archive file did exist, restore the backup elif archive_file_exists: FileHandler.move(src_file_path=backup_archive_path, dst_file_path=archive_path) # Clear info json files if true if clear_info_json_files: info_json_files = [ Path(self.working_directory) / path for path in os.listdir(self.working_directory) if path.endswith(".info.json") ] for info_json_file in info_json_files: FileHandler.delete(info_json_file) def _extract_entry_info_with_retry(self, entry: Entry) -> Entry: download_entry_dict = YTDLP.extract_info_with_retry( ytdl_options_overrides=self.download_ytdl_options, is_downloaded_fn=None if self.is_dry_run else entry.is_downloaded, is_thumbnail_downloaded_fn=None if (self.is_dry_run or not self.is_entry_thumbnails_enabled) else entry.is_thumbnail_downloaded, url=entry.webpage_url, ) return Entry(download_entry_dict, working_directory=self.working_directory) def _iterate_child_entries( self, url_validator: UrlValidator, entries: List[Entry] ) -> Iterator[Entry]: entries_to_iterate = entries if url_validator.download_reverse: entries_to_iterate = reversed(entries) for entry in entries_to_iterate: self._url_state.entries_downloaded += 1 if self._is_downloaded(entry): download_logger.info( "Already downloaded entry %d/%d: %s", self._url_state.entries_downloaded, self._url_state.entries_total, entry.title, ) continue yield entry self._mark_downloaded(entry) def _iterate_parent_entry( self, url_validator: UrlValidator, parent: EntryParent ) -> Iterator[Entry]: for entry_child in self._iterate_child_entries( url_validator=url_validator, entries=parent.entry_children() ): yield entry_child # Recursion the parent's parent entries for parent_child in reversed(parent.parent_children()): for entry_child in self._iterate_parent_entry( url_validator=url_validator, parent=parent_child ): yield entry_child def _download_url_metadata( self, collection_url: UrlValidator ) -> Tuple[List[EntryParent], List[Entry]]: """ Downloads only info.json files and forms EntryParent trees """ url = self.overrides.apply_formatter(collection_url.url) with self._separate_download_archives(): entry_dicts = YTDLP.extract_info_via_info_json( working_directory=self.working_directory, ytdl_options_overrides=self.metadata_ytdl_options, log_prefix_on_info_json_dl="Downloading metadata for", url=url, ) parents = EntryParent.from_entry_dicts( url=url, entry_dicts=entry_dicts, working_directory=self.working_directory, ) orphans = EntryParent.from_entry_dicts_with_no_parents( parents=parents, entry_dicts=entry_dicts, working_directory=self.working_directory ) return parents, orphans def _iterate_entries( self, url_validator: UrlValidator, parents: List[EntryParent], orphans: List[Entry], ) -> Iterator[Entry]: """ Downloads the leaf entries from EntryParent trees """ # Delete info json files afterwards so other collection URLs do not use them with self._separate_download_archives(clear_info_json_files=True): for parent in parents: for entry_child in self._iterate_parent_entry( url_validator=url_validator, parent=parent ): yield entry_child for orphan in self._iterate_child_entries(url_validator=url_validator, entries=orphans): yield orphan def download_metadata(self) -> Iterable[Entry]: """The function to perform the download of all media entries""" # download the bottom-most urls first since they are top-priority for collection_url in reversed(self.collection.urls.list): parents, orphan_entries = self._download_url_metadata(collection_url=collection_url) # TODO: Encapsulate this logic into its own class self._url_state = URLDownloadState( entries_total=sum(parent.num_children() for parent in parents) + len(orphan_entries) ) download_logger.info( "Beginning downloads for %s", self.overrides.apply_formatter(collection_url.url) ) for entry in self._iterate_entries( url_validator=collection_url, parents=parents, orphans=orphan_entries ): # Add the collection URL to the info_dict to trace where it came from entry.add_kwargs( {COLLECTION_URL: self.overrides.apply_formatter(collection_url.url)} ) yield entry def download(self, entry: Entry) -> Entry: """ Parameters ---------- entry Entry to download Returns ------- The entry that was downloaded successfully """ download_logger.info( "Downloading entry %d/%d: %s", self._url_state.entries_downloaded, self._url_state.entries_total, entry.title, ) download_entry = self._extract_entry_info_with_retry(entry=entry) upload_date_idx = self._enhanced_download_archive.mapping.get_num_entries_with_upload_date( upload_date_standardized=entry.upload_date_standardized ) download_idx = self._enhanced_download_archive.num_entries entry.add_kwargs( { # Subtitles are not downloaded in metadata run, only here, so move over REQUESTED_SUBTITLES: download_entry.kwargs_get(REQUESTED_SUBTITLES), # Same with sponsorblock chapters SPONSORBLOCK_CHAPTERS: download_entry.kwargs_get(SPONSORBLOCK_CHAPTERS), COMMENTS: download_entry.kwargs_get(COMMENTS), # Tracks number of entries downloaded DOWNLOAD_INDEX: download_idx, # Tracks number of entries with the same upload date to make them unique UPLOAD_DATE_INDEX: upload_date_idx, } ) return entry