Use collection download for YT channel and playlist
This commit is contained in:
parent
f741649001
commit
252f0789b8
6 changed files with 60 additions and 176 deletions
|
|
@ -204,37 +204,6 @@ class Downloader(DownloadArchiver, Generic[DownloaderOptionsT, DownloaderEntryT]
|
|||
f"yt-dlp failed to download an entry with these arguments: {error_dict}"
|
||||
)
|
||||
|
||||
def _filter_entry_dicts(
|
||||
self,
|
||||
entry_dicts: List[Dict],
|
||||
extractor: Optional[str] = None,
|
||||
sort_by: Optional[str] = None,
|
||||
) -> List[Dict]:
|
||||
"""
|
||||
Parameters
|
||||
----------
|
||||
entry_dicts
|
||||
entry dicts to filter
|
||||
extractor
|
||||
Optional. Extractor that the entry dicts must have. If None, defaults to the
|
||||
entry type's extractor
|
||||
sort_by
|
||||
Optional. Sort the entry dicts on this key
|
||||
|
||||
Returns
|
||||
-------
|
||||
filtered entry dicts
|
||||
"""
|
||||
if extractor is None:
|
||||
extractor = self.downloader_entry_type.entry_extractor
|
||||
|
||||
output = [
|
||||
entry_dict for entry_dict in entry_dicts if entry_dict.get("extractor") == extractor
|
||||
]
|
||||
if sort_by:
|
||||
output = sorted(output, key=lambda entry_dict: entry_dict[sort_by])
|
||||
return output
|
||||
|
||||
def _get_entry_dicts_from_info_json_files(self) -> List[Dict]:
|
||||
"""
|
||||
Returns
|
||||
|
|
@ -266,7 +235,6 @@ class Downloader(DownloadArchiver, Generic[DownloaderOptionsT, DownloaderEntryT]
|
|||
|
||||
info_json_listener = LogEntriesDownloadedListener(
|
||||
working_directory=self.working_directory,
|
||||
info_json_extractor=self.downloader_entry_type.entry_extractor,
|
||||
log_prefix=log_prefix,
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -246,18 +246,28 @@ class CollectionDownloader(Downloader[CollectionDownloadOptions, Entry]):
|
|||
for entry_child in self._download_parent_entry(parent=parent_child):
|
||||
yield entry_child
|
||||
|
||||
def _download_collection_url(
|
||||
self, collection_url: CollectionUrlValidator
|
||||
) -> Generator[Entry, None, None]:
|
||||
def download_url_metadata(self, collection_url: CollectionUrlValidator) -> List[EntryParent]:
|
||||
"""
|
||||
Downloads only info.json files and forms EntryParent trees
|
||||
"""
|
||||
with self._separate_download_archives():
|
||||
entry_dicts = self.extract_info_via_info_json(
|
||||
only_info_json=True,
|
||||
url=collection_url.url,
|
||||
log_prefix_on_info_json_dl="Downloading metadata for",
|
||||
)
|
||||
|
||||
parents = EntryParent.from_entry_dicts(
|
||||
entry_dicts=entry_dicts, working_directory=self.working_directory
|
||||
)
|
||||
return EntryParent.from_entry_dicts(
|
||||
entry_dicts=entry_dicts, working_directory=self.working_directory
|
||||
)
|
||||
|
||||
def download_url(
|
||||
self, collection_url: CollectionUrlValidator, parents: List[EntryParent]
|
||||
) -> Generator[Entry, None, None]:
|
||||
"""
|
||||
Downloads the leaf entries from EntryParent trees
|
||||
"""
|
||||
with self._separate_download_archives():
|
||||
for parent in parents:
|
||||
for entry_child in self._download_parent_entry(parent=parent):
|
||||
entry_child.add_variables(
|
||||
|
|
@ -271,5 +281,6 @@ class CollectionDownloader(Downloader[CollectionDownloadOptions, Entry]):
|
|||
"""
|
||||
# download the bottom-most urls first since they are top-priority
|
||||
for collection_url in reversed(self.download_options.collection_urls.list):
|
||||
for entry in self._download_collection_url(collection_url=collection_url):
|
||||
parents = self.download_url_metadata(collection_url=collection_url)
|
||||
for entry in self.download_url(collection_url=collection_url, parents=parents):
|
||||
yield entry
|
||||
|
|
|
|||
|
|
@ -7,14 +7,14 @@ from typing import Optional
|
|||
from ytdl_sub.config.preset_options import Overrides
|
||||
from ytdl_sub.downloaders.downloader import DownloaderOptionsT
|
||||
from ytdl_sub.downloaders.downloader import download_logger
|
||||
from ytdl_sub.downloaders.generic.collection import CollectionDownloader
|
||||
from ytdl_sub.downloaders.generic.collection import CollectionDownloadOptions
|
||||
from ytdl_sub.downloaders.youtube.abc import YoutubeDownloader
|
||||
from ytdl_sub.downloaders.youtube.abc import YoutubeDownloaderOptions
|
||||
from ytdl_sub.downloaders.ytdl_options_builder import YTDLOptionsBuilder
|
||||
from ytdl_sub.entries.entry_parent import EntryParent
|
||||
from ytdl_sub.entries.youtube import YoutubeVideo
|
||||
from ytdl_sub.utils.datetime import to_date_range_hack
|
||||
from ytdl_sub.utils.thumbnail import convert_url_thumbnail
|
||||
from ytdl_sub.validators.string_datetime import StringDatetimeValidator
|
||||
from ytdl_sub.validators.string_formatter_validators import OverridesStringFormatterValidator
|
||||
from ytdl_sub.validators.url_validator import YoutubeChannelUrlValidator
|
||||
from ytdl_sub.ytdl_additions.enhanced_download_archive import EnhancedDownloadArchive
|
||||
|
|
@ -41,8 +41,6 @@ class YoutubeChannelDownloaderOptions(YoutubeDownloaderOptions):
|
|||
|
||||
_required_keys = {"channel_url"}
|
||||
_optional_keys = {
|
||||
"before",
|
||||
"after",
|
||||
"channel_avatar_path",
|
||||
"channel_banner_path",
|
||||
}
|
||||
|
|
@ -58,8 +56,11 @@ class YoutubeChannelDownloaderOptions(YoutubeDownloaderOptions):
|
|||
self._channel_banner_path = self._validate_key_if_present(
|
||||
"channel_banner_path", OverridesStringFormatterValidator
|
||||
)
|
||||
self._before = self._validate_key_if_present("before", StringDatetimeValidator)
|
||||
self._after = self._validate_key_if_present("after", StringDatetimeValidator)
|
||||
|
||||
self.collection_validator = CollectionDownloadOptions(
|
||||
name=self._name,
|
||||
value={"urls": [{"url": self._channel_url}]},
|
||||
)
|
||||
|
||||
@property
|
||||
def channel_url(self) -> str:
|
||||
|
|
@ -84,22 +85,6 @@ class YoutubeChannelDownloaderOptions(YoutubeDownloaderOptions):
|
|||
"""
|
||||
return self._channel_banner_path
|
||||
|
||||
@property
|
||||
def before(self) -> Optional[StringDatetimeValidator]:
|
||||
"""
|
||||
DEPRECATED: use the `date_range` plugin instead. Will be removed in version 0.5.0
|
||||
Optional. Only download videos before this datetime.
|
||||
"""
|
||||
return self._before
|
||||
|
||||
@property
|
||||
def after(self) -> Optional[StringDatetimeValidator]:
|
||||
"""
|
||||
DEPRECATED: use the `date_range` plugin instead. Will be removed in version 0.5.0
|
||||
Optional. Only download videos after this datetime.
|
||||
"""
|
||||
return self._after
|
||||
|
||||
|
||||
class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions, YoutubeVideo]):
|
||||
downloader_options_type = YoutubeChannelDownloaderOptions
|
||||
|
|
@ -162,31 +147,19 @@ class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions
|
|||
"""
|
||||
Downloads all videos from a channel
|
||||
"""
|
||||
ytdl_options_overrides = {}
|
||||
|
||||
# If a date range is specified when download a YT channel, add it into the ytdl options
|
||||
source_date_range = to_date_range_hack(
|
||||
before=self.download_options.before, after=self.download_options.after
|
||||
downloader = CollectionDownloader(
|
||||
download_options=self.download_options.collection_validator,
|
||||
enhanced_download_archive=self._enhanced_download_archive,
|
||||
ytdl_options_builder=self._ytdl_options_builder,
|
||||
overrides=self.overrides,
|
||||
)
|
||||
if source_date_range:
|
||||
ytdl_options_overrides["daterange"] = source_date_range
|
||||
collection_url = self.download_options.collection_validator.collection_urls.list[0]
|
||||
|
||||
# dry-run the entire channel download first, this will get the
|
||||
# videos that will be downloaded. Afterwards, download each video one-by-one
|
||||
entry_dicts = self.extract_info_via_info_json(
|
||||
ytdl_options_overrides=ytdl_options_overrides,
|
||||
only_info_json=True,
|
||||
log_prefix_on_info_json_dl="Downloading metadata for",
|
||||
url=self.download_options.channel_url,
|
||||
)
|
||||
|
||||
parents: List[EntryParent] = EntryParent.from_entry_dicts(
|
||||
entry_dicts=entry_dicts,
|
||||
working_directory=self.working_directory,
|
||||
)
|
||||
assert len(parents) == 1, "Channel should be the only parent"
|
||||
parents = downloader.download_url_metadata(collection_url=collection_url)
|
||||
assert len(parents) == 1, "Channel should be the only entry parent"
|
||||
self.channel = parents[0]
|
||||
|
||||
# TODO: Handle this better
|
||||
self.overrides.add_override_variables(
|
||||
variables_to_add={
|
||||
"source_uploader": self.channel.kwargs_get("uploader", "__failed_to_scrape__"),
|
||||
|
|
@ -195,28 +168,11 @@ class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions
|
|||
}
|
||||
)
|
||||
|
||||
# Iterate in descending order to process older videos first. In case an error occurs and a
|
||||
# the channel must be redownloaded, it will fetch most recent metadata first, and break
|
||||
# on the older video that's been processed and is in the download archive.
|
||||
for idx, video in enumerate(reversed(self.channel.child_entries), start=1):
|
||||
video = video.to_type(YoutubeVideo)
|
||||
download_logger.info("Downloading %d/%d %s", idx, self.channel.child_count, video.title)
|
||||
|
||||
# Re-download the contents even if it's a dry-run as a single video. At this time,
|
||||
# channels do not download subtitles or subtitle metadata
|
||||
as_single_video_dict = self.extract_info_with_retry(
|
||||
is_downloaded_fn=None if self.is_dry_run else video.is_downloaded,
|
||||
ytdl_options_overrides={"writeinfojson": False, "skip_download": self.is_dry_run},
|
||||
url=video.webpage_url,
|
||||
)
|
||||
|
||||
# Workaround for the ytdlp issue
|
||||
for entry in downloader.download_url(collection_url=collection_url, parents=parents):
|
||||
# pylint: disable=protected-access
|
||||
video._kwargs["requested_subtitles"] = as_single_video_dict.get("requested_subtitles")
|
||||
yield YoutubeVideo(entry_dict=entry._kwargs, working_directory=self.working_directory)
|
||||
# pylint: enable=protected-access
|
||||
|
||||
yield video
|
||||
|
||||
def _download_thumbnail(
|
||||
self,
|
||||
thumbnail_url: str,
|
||||
|
|
|
|||
|
|
@ -2,10 +2,10 @@ from typing import Dict
|
|||
from typing import Generator
|
||||
from typing import List
|
||||
|
||||
from ytdl_sub.downloaders.downloader import download_logger
|
||||
from ytdl_sub.downloaders.generic.collection import CollectionDownloader
|
||||
from ytdl_sub.downloaders.generic.collection import CollectionDownloadOptions
|
||||
from ytdl_sub.downloaders.youtube.abc import YoutubeDownloader
|
||||
from ytdl_sub.downloaders.youtube.abc import YoutubeDownloaderOptions
|
||||
from ytdl_sub.entries.entry_parent import EntryParent
|
||||
from ytdl_sub.entries.youtube import YoutubePlaylistVideo
|
||||
from ytdl_sub.validators.url_validator import YoutubePlaylistUrlValidator
|
||||
|
||||
|
|
@ -34,6 +34,11 @@ class YoutubePlaylistDownloaderOptions(YoutubeDownloaderOptions):
|
|||
"playlist_url", YoutubePlaylistUrlValidator
|
||||
).playlist_url
|
||||
|
||||
self.collection_validator = CollectionDownloadOptions(
|
||||
name=self._name,
|
||||
value={"urls": [{"url": self._playlist_url}]},
|
||||
)
|
||||
|
||||
@property
|
||||
def playlist_url(self) -> str:
|
||||
"""
|
||||
|
|
@ -85,23 +90,20 @@ class YoutubePlaylistDownloader(
|
|||
def download(self) -> Generator[YoutubePlaylistVideo, None, None]:
|
||||
"""
|
||||
Downloads all videos in a Youtube playlist.
|
||||
|
||||
Dry-run the entire playlist download first. This will get the videos that will be
|
||||
downloaded. Afterwards, download each video one-by-one
|
||||
"""
|
||||
entry_dicts = self.extract_info_via_info_json(
|
||||
only_info_json=True,
|
||||
log_prefix_on_info_json_dl="Downloading metadata for",
|
||||
url=self.download_options.playlist_url,
|
||||
downloader = CollectionDownloader(
|
||||
download_options=self.download_options.collection_validator,
|
||||
enhanced_download_archive=self._enhanced_download_archive,
|
||||
ytdl_options_builder=self._ytdl_options_builder,
|
||||
overrides=self.overrides,
|
||||
)
|
||||
collection_url = self.download_options.collection_validator.collection_urls.list[0]
|
||||
|
||||
parents: List[EntryParent] = EntryParent.from_entry_dicts(
|
||||
entry_dicts=entry_dicts,
|
||||
working_directory=self.working_directory,
|
||||
)
|
||||
assert len(parents) == 1, "Playlist should be the only parent"
|
||||
parents = downloader.download_url_metadata(collection_url=collection_url)
|
||||
assert len(parents) == 1, "Playlist should be the only entry parent"
|
||||
playlist = parents[0]
|
||||
|
||||
# TODO: Handle this better
|
||||
self.overrides.add_override_variables(
|
||||
variables_to_add={
|
||||
"source_title": playlist.title,
|
||||
|
|
@ -110,24 +112,9 @@ class YoutubePlaylistDownloader(
|
|||
}
|
||||
)
|
||||
|
||||
# Iterate in reverse order to process older videos first. In case an error occurs and a
|
||||
# the playlist must be redownloaded, it will fetch most recent metadata first, and break
|
||||
# on the older video that's been processed and is in the download archive.
|
||||
for idx, video in enumerate(reversed(playlist.child_entries), start=1):
|
||||
video = video.to_type(YoutubePlaylistVideo)
|
||||
download_logger.info("Downloading %d/%d %s", idx, len(entry_dicts), video.title)
|
||||
|
||||
# Re-download the contents even if it's a dry-run as a single video. At this time,
|
||||
# playlists do not download subtitles or subtitle metadata
|
||||
as_single_video_dict = self.extract_info_with_retry(
|
||||
is_downloaded_fn=None if self.is_dry_run else video.is_downloaded,
|
||||
ytdl_options_overrides={"writeinfojson": False, "skip_download": self.is_dry_run},
|
||||
url=video.webpage_url,
|
||||
)
|
||||
|
||||
# Workaround for the ytdlp issue
|
||||
for entry in downloader.download_url(collection_url=collection_url, parents=parents):
|
||||
# pylint: disable=protected-access
|
||||
video._kwargs["requested_subtitles"] = as_single_video_dict.get("requested_subtitles")
|
||||
yield YoutubePlaylistVideo(
|
||||
entry_dict=entry._kwargs, working_directory=self.working_directory
|
||||
)
|
||||
# pylint: enable=protected-access
|
||||
|
||||
yield video
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ logger = Logger.get(name="downloader")
|
|||
|
||||
|
||||
class LogEntriesDownloadedListener(threading.Thread):
|
||||
def __init__(self, working_directory: str, info_json_extractor: str, log_prefix: str):
|
||||
def __init__(self, working_directory: str, log_prefix: str):
|
||||
"""
|
||||
To be ran in a thread while download via ytdl-sub. Listens for new .info.json files in the
|
||||
working directory, checks the extractor value, and if it matches the input arg, log the
|
||||
|
|
@ -23,20 +23,18 @@ class LogEntriesDownloadedListener(threading.Thread):
|
|||
----------
|
||||
working_directory
|
||||
subscription download working directory
|
||||
info_json_extractor
|
||||
print the titles of the info.json file with this extractor
|
||||
log_prefix
|
||||
The message to print prefixed to the title, i.e. '{log_prefix} {title}'
|
||||
"""
|
||||
threading.Thread.__init__(self)
|
||||
self.working_directory = working_directory
|
||||
self.info_json_extractor = info_json_extractor
|
||||
self.log_prefix = log_prefix
|
||||
self.complete = False
|
||||
|
||||
self._files_read: Set[str] = set()
|
||||
|
||||
def _get_title_from_info_json(self, path: Path) -> Optional[str]:
|
||||
@classmethod
|
||||
def _get_title_from_info_json(cls, path: Path) -> Optional[str]:
|
||||
try:
|
||||
with open(path, "r", encoding="utf-8") as file:
|
||||
file_json = json.load(file)
|
||||
|
|
@ -44,10 +42,7 @@ class LogEntriesDownloadedListener(threading.Thread):
|
|||
# swallow the error since this is only printing logs
|
||||
return None
|
||||
|
||||
if file_json.get("extractor") == self.info_json_extractor:
|
||||
return file_json.get("title")
|
||||
|
||||
return None
|
||||
return file_json.get("title")
|
||||
|
||||
@classmethod
|
||||
def _is_info_json(cls, path: Path) -> bool:
|
||||
|
|
|
|||
|
|
@ -3,42 +3,9 @@ from typing import Optional
|
|||
from yt_dlp import DateRange
|
||||
|
||||
from ytdl_sub.config.preset_options import Overrides
|
||||
from ytdl_sub.utils.logger import Logger
|
||||
from ytdl_sub.validators.string_datetime import StringDatetimeValidator
|
||||
|
||||
|
||||
def to_date_range_hack(
|
||||
before: Optional[StringDatetimeValidator], after: Optional[StringDatetimeValidator]
|
||||
) -> Optional[DateRange]:
|
||||
"""
|
||||
Workaround for channel before/after support.
|
||||
|
||||
Returns
|
||||
-------
|
||||
Date range if the 'before' or 'after' is defined. None otherwise.
|
||||
"""
|
||||
start: Optional[str] = None
|
||||
end: Optional[str] = None
|
||||
|
||||
if after:
|
||||
start = after.apply_formatter(variable_dict={})
|
||||
|
||||
if before:
|
||||
end = before.apply_formatter(variable_dict={})
|
||||
|
||||
if start or end:
|
||||
logger = Logger.get(name="youtube-channel")
|
||||
logger.warning(
|
||||
"DEPRECATED: youtube.before/after will are deprecated and will be removed in v0.0.5. "
|
||||
"Use the 'date_range' plugin instead: "
|
||||
"https://ytdl-sub.readthedocs.io/en/latest/config.html#date-range"
|
||||
)
|
||||
|
||||
return DateRange(start=start, end=end)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def to_date_range(
|
||||
before: Optional[StringDatetimeValidator],
|
||||
after: Optional[StringDatetimeValidator],
|
||||
|
|
|
|||
Loading…
Reference in a new issue