diff --git a/src/ytdl_sub/downloaders/downloader.py b/src/ytdl_sub/downloaders/downloader.py index 7b966a22..e4c67b1f 100644 --- a/src/ytdl_sub/downloaders/downloader.py +++ b/src/ytdl_sub/downloaders/downloader.py @@ -204,37 +204,6 @@ class Downloader(DownloadArchiver, Generic[DownloaderOptionsT, DownloaderEntryT] f"yt-dlp failed to download an entry with these arguments: {error_dict}" ) - def _filter_entry_dicts( - self, - entry_dicts: List[Dict], - extractor: Optional[str] = None, - sort_by: Optional[str] = None, - ) -> List[Dict]: - """ - Parameters - ---------- - entry_dicts - entry dicts to filter - extractor - Optional. Extractor that the entry dicts must have. If None, defaults to the - entry type's extractor - sort_by - Optional. Sort the entry dicts on this key - - Returns - ------- - filtered entry dicts - """ - if extractor is None: - extractor = self.downloader_entry_type.entry_extractor - - output = [ - entry_dict for entry_dict in entry_dicts if entry_dict.get("extractor") == extractor - ] - if sort_by: - output = sorted(output, key=lambda entry_dict: entry_dict[sort_by]) - return output - def _get_entry_dicts_from_info_json_files(self) -> List[Dict]: """ Returns @@ -266,7 +235,6 @@ class Downloader(DownloadArchiver, Generic[DownloaderOptionsT, DownloaderEntryT] info_json_listener = LogEntriesDownloadedListener( working_directory=self.working_directory, - info_json_extractor=self.downloader_entry_type.entry_extractor, log_prefix=log_prefix, ) diff --git a/src/ytdl_sub/downloaders/generic/collection.py b/src/ytdl_sub/downloaders/generic/collection.py index e1058d9f..64157d0d 100644 --- a/src/ytdl_sub/downloaders/generic/collection.py +++ b/src/ytdl_sub/downloaders/generic/collection.py @@ -246,18 +246,28 @@ class CollectionDownloader(Downloader[CollectionDownloadOptions, Entry]): for entry_child in self._download_parent_entry(parent=parent_child): yield entry_child - def _download_collection_url( - self, collection_url: CollectionUrlValidator - ) -> Generator[Entry, None, None]: + def download_url_metadata(self, collection_url: CollectionUrlValidator) -> List[EntryParent]: + """ + Downloads only info.json files and forms EntryParent trees + """ with self._separate_download_archives(): entry_dicts = self.extract_info_via_info_json( only_info_json=True, url=collection_url.url, + log_prefix_on_info_json_dl="Downloading metadata for", ) - parents = EntryParent.from_entry_dicts( - entry_dicts=entry_dicts, working_directory=self.working_directory - ) + return EntryParent.from_entry_dicts( + entry_dicts=entry_dicts, working_directory=self.working_directory + ) + + def download_url( + self, collection_url: CollectionUrlValidator, parents: List[EntryParent] + ) -> Generator[Entry, None, None]: + """ + Downloads the leaf entries from EntryParent trees + """ + with self._separate_download_archives(): for parent in parents: for entry_child in self._download_parent_entry(parent=parent): entry_child.add_variables( @@ -271,5 +281,6 @@ class CollectionDownloader(Downloader[CollectionDownloadOptions, Entry]): """ # download the bottom-most urls first since they are top-priority for collection_url in reversed(self.download_options.collection_urls.list): - for entry in self._download_collection_url(collection_url=collection_url): + parents = self.download_url_metadata(collection_url=collection_url) + for entry in self.download_url(collection_url=collection_url, parents=parents): yield entry diff --git a/src/ytdl_sub/downloaders/youtube/channel.py b/src/ytdl_sub/downloaders/youtube/channel.py index 08e38cb0..4e581e7b 100644 --- a/src/ytdl_sub/downloaders/youtube/channel.py +++ b/src/ytdl_sub/downloaders/youtube/channel.py @@ -7,14 +7,14 @@ from typing import Optional from ytdl_sub.config.preset_options import Overrides from ytdl_sub.downloaders.downloader import DownloaderOptionsT from ytdl_sub.downloaders.downloader import download_logger +from ytdl_sub.downloaders.generic.collection import CollectionDownloader +from ytdl_sub.downloaders.generic.collection import CollectionDownloadOptions from ytdl_sub.downloaders.youtube.abc import YoutubeDownloader from ytdl_sub.downloaders.youtube.abc import YoutubeDownloaderOptions from ytdl_sub.downloaders.ytdl_options_builder import YTDLOptionsBuilder from ytdl_sub.entries.entry_parent import EntryParent from ytdl_sub.entries.youtube import YoutubeVideo -from ytdl_sub.utils.datetime import to_date_range_hack from ytdl_sub.utils.thumbnail import convert_url_thumbnail -from ytdl_sub.validators.string_datetime import StringDatetimeValidator from ytdl_sub.validators.string_formatter_validators import OverridesStringFormatterValidator from ytdl_sub.validators.url_validator import YoutubeChannelUrlValidator from ytdl_sub.ytdl_additions.enhanced_download_archive import EnhancedDownloadArchive @@ -41,8 +41,6 @@ class YoutubeChannelDownloaderOptions(YoutubeDownloaderOptions): _required_keys = {"channel_url"} _optional_keys = { - "before", - "after", "channel_avatar_path", "channel_banner_path", } @@ -58,8 +56,11 @@ class YoutubeChannelDownloaderOptions(YoutubeDownloaderOptions): self._channel_banner_path = self._validate_key_if_present( "channel_banner_path", OverridesStringFormatterValidator ) - self._before = self._validate_key_if_present("before", StringDatetimeValidator) - self._after = self._validate_key_if_present("after", StringDatetimeValidator) + + self.collection_validator = CollectionDownloadOptions( + name=self._name, + value={"urls": [{"url": self._channel_url}]}, + ) @property def channel_url(self) -> str: @@ -84,22 +85,6 @@ class YoutubeChannelDownloaderOptions(YoutubeDownloaderOptions): """ return self._channel_banner_path - @property - def before(self) -> Optional[StringDatetimeValidator]: - """ - DEPRECATED: use the `date_range` plugin instead. Will be removed in version 0.5.0 - Optional. Only download videos before this datetime. - """ - return self._before - - @property - def after(self) -> Optional[StringDatetimeValidator]: - """ - DEPRECATED: use the `date_range` plugin instead. Will be removed in version 0.5.0 - Optional. Only download videos after this datetime. - """ - return self._after - class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions, YoutubeVideo]): downloader_options_type = YoutubeChannelDownloaderOptions @@ -162,31 +147,19 @@ class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions """ Downloads all videos from a channel """ - ytdl_options_overrides = {} - - # If a date range is specified when download a YT channel, add it into the ytdl options - source_date_range = to_date_range_hack( - before=self.download_options.before, after=self.download_options.after + downloader = CollectionDownloader( + download_options=self.download_options.collection_validator, + enhanced_download_archive=self._enhanced_download_archive, + ytdl_options_builder=self._ytdl_options_builder, + overrides=self.overrides, ) - if source_date_range: - ytdl_options_overrides["daterange"] = source_date_range + collection_url = self.download_options.collection_validator.collection_urls.list[0] - # dry-run the entire channel download first, this will get the - # videos that will be downloaded. Afterwards, download each video one-by-one - entry_dicts = self.extract_info_via_info_json( - ytdl_options_overrides=ytdl_options_overrides, - only_info_json=True, - log_prefix_on_info_json_dl="Downloading metadata for", - url=self.download_options.channel_url, - ) - - parents: List[EntryParent] = EntryParent.from_entry_dicts( - entry_dicts=entry_dicts, - working_directory=self.working_directory, - ) - assert len(parents) == 1, "Channel should be the only parent" + parents = downloader.download_url_metadata(collection_url=collection_url) + assert len(parents) == 1, "Channel should be the only entry parent" self.channel = parents[0] + # TODO: Handle this better self.overrides.add_override_variables( variables_to_add={ "source_uploader": self.channel.kwargs_get("uploader", "__failed_to_scrape__"), @@ -195,28 +168,11 @@ class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions } ) - # Iterate in descending order to process older videos first. In case an error occurs and a - # the channel must be redownloaded, it will fetch most recent metadata first, and break - # on the older video that's been processed and is in the download archive. - for idx, video in enumerate(reversed(self.channel.child_entries), start=1): - video = video.to_type(YoutubeVideo) - download_logger.info("Downloading %d/%d %s", idx, self.channel.child_count, video.title) - - # Re-download the contents even if it's a dry-run as a single video. At this time, - # channels do not download subtitles or subtitle metadata - as_single_video_dict = self.extract_info_with_retry( - is_downloaded_fn=None if self.is_dry_run else video.is_downloaded, - ytdl_options_overrides={"writeinfojson": False, "skip_download": self.is_dry_run}, - url=video.webpage_url, - ) - - # Workaround for the ytdlp issue + for entry in downloader.download_url(collection_url=collection_url, parents=parents): # pylint: disable=protected-access - video._kwargs["requested_subtitles"] = as_single_video_dict.get("requested_subtitles") + yield YoutubeVideo(entry_dict=entry._kwargs, working_directory=self.working_directory) # pylint: enable=protected-access - yield video - def _download_thumbnail( self, thumbnail_url: str, diff --git a/src/ytdl_sub/downloaders/youtube/playlist.py b/src/ytdl_sub/downloaders/youtube/playlist.py index 6b0d03f9..b81a66ce 100644 --- a/src/ytdl_sub/downloaders/youtube/playlist.py +++ b/src/ytdl_sub/downloaders/youtube/playlist.py @@ -2,10 +2,10 @@ from typing import Dict from typing import Generator from typing import List -from ytdl_sub.downloaders.downloader import download_logger +from ytdl_sub.downloaders.generic.collection import CollectionDownloader +from ytdl_sub.downloaders.generic.collection import CollectionDownloadOptions from ytdl_sub.downloaders.youtube.abc import YoutubeDownloader from ytdl_sub.downloaders.youtube.abc import YoutubeDownloaderOptions -from ytdl_sub.entries.entry_parent import EntryParent from ytdl_sub.entries.youtube import YoutubePlaylistVideo from ytdl_sub.validators.url_validator import YoutubePlaylistUrlValidator @@ -34,6 +34,11 @@ class YoutubePlaylistDownloaderOptions(YoutubeDownloaderOptions): "playlist_url", YoutubePlaylistUrlValidator ).playlist_url + self.collection_validator = CollectionDownloadOptions( + name=self._name, + value={"urls": [{"url": self._playlist_url}]}, + ) + @property def playlist_url(self) -> str: """ @@ -85,23 +90,20 @@ class YoutubePlaylistDownloader( def download(self) -> Generator[YoutubePlaylistVideo, None, None]: """ Downloads all videos in a Youtube playlist. - - Dry-run the entire playlist download first. This will get the videos that will be - downloaded. Afterwards, download each video one-by-one """ - entry_dicts = self.extract_info_via_info_json( - only_info_json=True, - log_prefix_on_info_json_dl="Downloading metadata for", - url=self.download_options.playlist_url, + downloader = CollectionDownloader( + download_options=self.download_options.collection_validator, + enhanced_download_archive=self._enhanced_download_archive, + ytdl_options_builder=self._ytdl_options_builder, + overrides=self.overrides, ) + collection_url = self.download_options.collection_validator.collection_urls.list[0] - parents: List[EntryParent] = EntryParent.from_entry_dicts( - entry_dicts=entry_dicts, - working_directory=self.working_directory, - ) - assert len(parents) == 1, "Playlist should be the only parent" + parents = downloader.download_url_metadata(collection_url=collection_url) + assert len(parents) == 1, "Playlist should be the only entry parent" playlist = parents[0] + # TODO: Handle this better self.overrides.add_override_variables( variables_to_add={ "source_title": playlist.title, @@ -110,24 +112,9 @@ class YoutubePlaylistDownloader( } ) - # Iterate in reverse order to process older videos first. In case an error occurs and a - # the playlist must be redownloaded, it will fetch most recent metadata first, and break - # on the older video that's been processed and is in the download archive. - for idx, video in enumerate(reversed(playlist.child_entries), start=1): - video = video.to_type(YoutubePlaylistVideo) - download_logger.info("Downloading %d/%d %s", idx, len(entry_dicts), video.title) - - # Re-download the contents even if it's a dry-run as a single video. At this time, - # playlists do not download subtitles or subtitle metadata - as_single_video_dict = self.extract_info_with_retry( - is_downloaded_fn=None if self.is_dry_run else video.is_downloaded, - ytdl_options_overrides={"writeinfojson": False, "skip_download": self.is_dry_run}, - url=video.webpage_url, - ) - - # Workaround for the ytdlp issue + for entry in downloader.download_url(collection_url=collection_url, parents=parents): # pylint: disable=protected-access - video._kwargs["requested_subtitles"] = as_single_video_dict.get("requested_subtitles") + yield YoutubePlaylistVideo( + entry_dict=entry._kwargs, working_directory=self.working_directory + ) # pylint: enable=protected-access - - yield video diff --git a/src/ytdl_sub/thread/log_entries_downloaded_listener.py b/src/ytdl_sub/thread/log_entries_downloaded_listener.py index a20038b5..ffa9c3b9 100644 --- a/src/ytdl_sub/thread/log_entries_downloaded_listener.py +++ b/src/ytdl_sub/thread/log_entries_downloaded_listener.py @@ -13,7 +13,7 @@ logger = Logger.get(name="downloader") class LogEntriesDownloadedListener(threading.Thread): - def __init__(self, working_directory: str, info_json_extractor: str, log_prefix: str): + def __init__(self, working_directory: str, log_prefix: str): """ To be ran in a thread while download via ytdl-sub. Listens for new .info.json files in the working directory, checks the extractor value, and if it matches the input arg, log the @@ -23,20 +23,18 @@ class LogEntriesDownloadedListener(threading.Thread): ---------- working_directory subscription download working directory - info_json_extractor - print the titles of the info.json file with this extractor log_prefix The message to print prefixed to the title, i.e. '{log_prefix} {title}' """ threading.Thread.__init__(self) self.working_directory = working_directory - self.info_json_extractor = info_json_extractor self.log_prefix = log_prefix self.complete = False self._files_read: Set[str] = set() - def _get_title_from_info_json(self, path: Path) -> Optional[str]: + @classmethod + def _get_title_from_info_json(cls, path: Path) -> Optional[str]: try: with open(path, "r", encoding="utf-8") as file: file_json = json.load(file) @@ -44,10 +42,7 @@ class LogEntriesDownloadedListener(threading.Thread): # swallow the error since this is only printing logs return None - if file_json.get("extractor") == self.info_json_extractor: - return file_json.get("title") - - return None + return file_json.get("title") @classmethod def _is_info_json(cls, path: Path) -> bool: diff --git a/src/ytdl_sub/utils/datetime.py b/src/ytdl_sub/utils/datetime.py index 157629e1..d7b1e07a 100644 --- a/src/ytdl_sub/utils/datetime.py +++ b/src/ytdl_sub/utils/datetime.py @@ -3,42 +3,9 @@ from typing import Optional from yt_dlp import DateRange from ytdl_sub.config.preset_options import Overrides -from ytdl_sub.utils.logger import Logger from ytdl_sub.validators.string_datetime import StringDatetimeValidator -def to_date_range_hack( - before: Optional[StringDatetimeValidator], after: Optional[StringDatetimeValidator] -) -> Optional[DateRange]: - """ - Workaround for channel before/after support. - - Returns - ------- - Date range if the 'before' or 'after' is defined. None otherwise. - """ - start: Optional[str] = None - end: Optional[str] = None - - if after: - start = after.apply_formatter(variable_dict={}) - - if before: - end = before.apply_formatter(variable_dict={}) - - if start or end: - logger = Logger.get(name="youtube-channel") - logger.warning( - "DEPRECATED: youtube.before/after will are deprecated and will be removed in v0.0.5. " - "Use the 'date_range' plugin instead: " - "https://ytdl-sub.readthedocs.io/en/latest/config.html#date-range" - ) - - return DateRange(start=start, end=end) - - return None - - def to_date_range( before: Optional[StringDatetimeValidator], after: Optional[StringDatetimeValidator],