dry-run huge refactor

This commit is contained in:
jbannon 2022-06-29 17:06:31 +00:00
parent 4447f456d5
commit ab8fac47a8
12 changed files with 186 additions and 106 deletions

View file

@ -1,9 +1,9 @@
import argparse import argparse
from ytdl_sub.utils.logger import LoggerLevels
################################################################################################### ###################################################################################################
# GLOBAL PARSER # GLOBAL PARSER
from ytdl_sub.utils.logger import LoggerLevels
parser = argparse.ArgumentParser( parser = argparse.ArgumentParser(
description="ytdl-sub: Automate download and adding metadata with YoutubeDL" description="ytdl-sub: Automate download and adding metadata with YoutubeDL"
) )
@ -15,6 +15,11 @@ parser.add_argument(
help="path to the config yaml, uses config.yaml if not provided", help="path to the config yaml, uses config.yaml if not provided",
default="config.yaml", default="config.yaml",
) )
parser.add_argument(
"--dry-run",
action="store_true",
help="does not perform any video downloads or writes to output directories",
)
parser.add_argument( parser.add_argument(
"--log-level", "--log-level",
metavar="|".join(LoggerLevels.names()), metavar="|".join(LoggerLevels.names()),

View file

@ -66,7 +66,6 @@ class Downloader(Generic[DownloaderOptionsT, DownloaderEntryT], ABC):
cls, cls,
working_directory: str, working_directory: str,
ytdl_options: Optional[Dict], ytdl_options: Optional[Dict],
download_archive_file_name: Optional[str],
) -> Dict: ) -> Dict:
"""Configure the ytdl options for the downloader""" """Configure the ytdl options for the downloader"""
if ytdl_options is None: if ytdl_options is None:
@ -81,12 +80,6 @@ class Downloader(Generic[DownloaderOptionsT, DownloaderEntryT], ABC):
# Overwrite the output location with the specified working directory # Overwrite the output location with the specified working directory
ytdl_options["outtmpl"] = str(Path(working_directory) / "%(id)s.%(ext)s") ytdl_options["outtmpl"] = str(Path(working_directory) / "%(id)s.%(ext)s")
# If a download archive file name is provided, set it to that
if download_archive_file_name:
ytdl_options["download_archive"] = str(
Path(working_directory) / download_archive_file_name
)
return ytdl_options return ytdl_options
def __init__( def __init__(
@ -94,7 +87,6 @@ class Downloader(Generic[DownloaderOptionsT, DownloaderEntryT], ABC):
working_directory: str, working_directory: str,
download_options: DownloaderOptionsT, download_options: DownloaderOptionsT,
ytdl_options: Optional[Dict] = None, ytdl_options: Optional[Dict] = None,
download_archive_file_name: Optional[str] = None,
): ):
""" """
Parameters Parameters
@ -105,15 +97,12 @@ class Downloader(Generic[DownloaderOptionsT, DownloaderEntryT], ABC):
Options validator for this downloader Options validator for this downloader
ytdl_options ytdl_options
YTDL options validator YTDL options validator
download_archive_file_name
Optional. Name of the download archive file that should reside in the working directory
""" """
self.working_directory = working_directory self.working_directory = working_directory
self.download_options = download_options self.download_options = download_options
self.ytdl_options = self._configure_ytdl_options( self.ytdl_options = self._configure_ytdl_options(
ytdl_options=ytdl_options, ytdl_options=ytdl_options,
working_directory=self.working_directory, working_directory=self.working_directory,
download_archive_file_name=download_archive_file_name,
) )
@contextmanager @contextmanager
@ -129,6 +118,15 @@ class Downloader(Generic[DownloaderOptionsT, DownloaderEntryT], ABC):
with ytdl.YoutubeDL(ytdl_options) as ytdl_downloader: with ytdl.YoutubeDL(ytdl_options) as ytdl_downloader:
yield ytdl_downloader yield ytdl_downloader
@property
def is_dry_run(self) -> bool:
"""
Returns
-------
True if dry-run is enabled. False otherwise.
"""
return self.ytdl_options.get("skip_download", False)
def extract_info(self, ytdl_options_overrides: Optional[Dict] = None, **kwargs) -> Dict: def extract_info(self, ytdl_options_overrides: Optional[Dict] = None, **kwargs) -> Dict:
""" """
Wrapper around yt_dlp.YoutubeDL.YoutubeDL.extract_info Wrapper around yt_dlp.YoutubeDL.YoutubeDL.extract_info

View file

@ -1,7 +1,6 @@
import copy import copy
import re import re
from pathlib import Path from pathlib import Path
from shutil import copyfile
from typing import Dict from typing import Dict
from typing import List from typing import List
@ -12,8 +11,6 @@ from ytdl_sub.entries.youtube import YoutubeVideo
from ytdl_sub.utils.chapters import Chapters from ytdl_sub.utils.chapters import Chapters
from ytdl_sub.utils.chapters import Timestamp from ytdl_sub.utils.chapters import Timestamp
from ytdl_sub.utils.ffmpeg import FFMPEG from ytdl_sub.utils.ffmpeg import FFMPEG
from ytdl_sub.utils.thumbnail import convert_download_thumbnail
from ytdl_sub.validators.validators import StringValidator
# Captures the following formats: # Captures the following formats:
# 0:00 title # 0:00 title
@ -21,6 +18,10 @@ from ytdl_sub.validators.validators import StringValidator
# 1:00:00 title # 1:00:00 title
# 01:00:00 title # 01:00:00 title
# where capture group 1 and 2 are the timestamp and title, respectively # where capture group 1 and 2 are the timestamp and title, respectively
from ytdl_sub.utils.file_handler import FileHandler
from ytdl_sub.utils.thumbnail import convert_download_thumbnail
from ytdl_sub.validators.validators import StringValidator
_SPLIT_TIMESTAMP_REGEX = re.compile(r"^((?:\d\d:)?(?:\d:)?(?:\d)?\d:\d\d) (.+)$") _SPLIT_TIMESTAMP_REGEX = re.compile(r"^((?:\d\d:)?(?:\d:)?(?:\d)?\d:\d\d) (.+)$")
@ -171,7 +172,10 @@ class YoutubeSplitVideoDownloader(
) )
) )
# Copy the thumbnail # Copy the thumbnail
copyfile(src=entry.get_download_thumbnail_path(), dst=output_thumbnail_file) FileHandler.copy(
src_file_path=entry.get_download_thumbnail_path(),
dst_file_path=output_thumbnail_file,
)
# Format the split video as a YoutubePlaylistVideo # Format the split video as a YoutubePlaylistVideo
split_videos.append( split_videos.append(

View file

@ -293,13 +293,11 @@ class YoutubeChannelDownloader(YoutubeDownloader[YoutubeChannelDownloaderOptions
working_directory: str, working_directory: str,
download_options: DownloaderOptionsT, download_options: DownloaderOptionsT,
ytdl_options: Optional[Dict] = None, ytdl_options: Optional[Dict] = None,
download_archive_file_name: Optional[str] = None,
): ):
super().__init__( super().__init__(
working_directory=working_directory, working_directory=working_directory,
download_options=download_options, download_options=download_options,
ytdl_options=ytdl_options, ytdl_options=ytdl_options,
download_archive_file_name=download_archive_file_name,
) )
self.channel: Optional[YoutubeChannel] = None self.channel: Optional[YoutubeChannel] = None

View file

@ -11,13 +11,29 @@ class Entry(EntryVariables, BaseEntry):
Entry object to represent a single media object returned from yt-dlp. Entry object to represent a single media object returned from yt-dlp.
""" """
def get_download_file_name(self) -> str:
"""
Returns
-------
The entry's file name
"""
return f"{self.uid}.{self.ext}"
def get_download_file_path(self) -> str: def get_download_file_path(self) -> str:
"""Returns the entry's file path to where it was downloaded""" """Returns the entry's file path to where it was downloaded"""
return str(Path(self.working_directory()) / f"{self.uid}.{self.ext}") return str(Path(self.working_directory()) / self.get_download_file_name())
def get_download_thumbnail_name(self) -> str:
"""
Returns
-------
The download thumbnail's file name
"""
return f"{self.uid}.{self.thumbnail_ext}"
def get_download_thumbnail_path(self) -> str: def get_download_thumbnail_path(self) -> str:
"""Returns the entry's thumbnail's file path to where it was downloaded""" """Returns the entry's thumbnail's file path to where it was downloaded"""
return str(Path(self.working_directory()) / f"{self.uid}.{self.thumbnail_ext}") return str(Path(self.working_directory()) / self.get_download_thumbnail_name())
@final @final
def to_dict(self) -> Dict[str, str]: def to_dict(self) -> Dict[str, str]:

View file

@ -109,10 +109,10 @@ class NfoTagsPlugin(Plugin[NfoTagsOptions]):
) )
# Save the nfo's XML to file # Save the nfo's XML to file
nfo_file_path = Path(self.output_directory) / nfo_file_name nfo_file_path = Path(self.working_directory) / nfo_file_name
os.makedirs(os.path.dirname(nfo_file_path), exist_ok=True) os.makedirs(os.path.dirname(nfo_file_path), exist_ok=True)
with open(nfo_file_path, "wb") as nfo_file: with open(nfo_file_path, "wb") as nfo_file:
nfo_file.write(xml) nfo_file.write(xml)
# Archive the nfo's file name # Archive the nfo's file name
self.archive_entry_file_name(entry=entry, relative_file_path=nfo_file_name) self.save_file(file_name=nfo_file_name, entry=entry)

View file

@ -100,7 +100,9 @@ class OutputDirectoryNfoTagsPlugin(Plugin[OutputDirectoryNfoTagsOptions]):
nfo_file_name = self.overrides.apply_formatter(formatter=self.plugin_options.nfo_name) nfo_file_name = self.overrides.apply_formatter(formatter=self.plugin_options.nfo_name)
# Save the nfo's XML to file # Save the nfo's XML to file
nfo_file_path = Path(self.output_directory) / nfo_file_name nfo_file_path = Path(self.working_directory) / nfo_file_name
os.makedirs(os.path.dirname(nfo_file_path), exist_ok=True) os.makedirs(os.path.dirname(nfo_file_path), exist_ok=True)
with open(nfo_file_path, "wb") as nfo_file: with open(nfo_file_path, "wb") as nfo_file:
nfo_file.write(xml) nfo_file.write(xml)
self.save_file(file_name=nfo_file_name)

View file

@ -32,33 +32,32 @@ class Plugin(Generic[PluginOptionsT], ABC):
def __init__( def __init__(
self, self,
plugin_options: PluginOptionsT, plugin_options: PluginOptionsT,
output_directory: str,
overrides: Overrides, overrides: Overrides,
enhanced_download_archive: Optional[EnhancedDownloadArchive], enhanced_download_archive: EnhancedDownloadArchive,
): ):
self.plugin_options = plugin_options self.plugin_options = plugin_options
self.output_directory = output_directory
self.overrides = overrides self.overrides = overrides
self.__enhanced_download_archive = enhanced_download_archive self.__enhanced_download_archive = enhanced_download_archive
# TODO pass yaml snake case name in the class somewhere, and use it for the logger # TODO pass yaml snake case name in the class somewhere, and use it for the logger
self._logger = Logger.get(self.__class__.__name__) self._logger = Logger.get(self.__class__.__name__)
@final @property
def archive_entry_file_name(self, entry: Entry, relative_file_path: str) -> None: def working_directory(self) -> str:
return self.__enhanced_download_archive.working_directory
def save_file(self, file_name: str, entry: Optional[Entry] = None) -> None:
""" """
Adds an entry and a file name that belongs to it into the archive mapping. Saves a file in the working directory to the output directory.
If maintain_download_archive is False for the subscription, this method will do nothing.
Parameters Parameters
---------- ----------
entry: file_name
Optional. The entry the file belongs to Name of the file relative to the working directory
relative_file_path: entry
The name of the file path relative to the output directory Optional. Entry that the file belongs to
""" """
if self.__enhanced_download_archive: self.__enhanced_download_archive.save_file(
self.__enhanced_download_archive.mapping.add_entry( file_name=file_name, output_file_name=file_name, entry=entry
entry=entry, entry_file_path=relative_file_path
) )
def post_process_entry(self, entry: Entry): def post_process_entry(self, entry: Entry):

View file

@ -1,8 +1,8 @@
import contextlib import contextlib
import copy
import os import os
import shutil import shutil
from pathlib import Path from pathlib import Path
from shutil import copyfile
from typing import List from typing import List
from typing import Tuple from typing import Tuple
from typing import Type from typing import Type
@ -144,29 +144,6 @@ class Subscription:
and self.downloader_class.supports_download_archive and self.downloader_class.supports_download_archive
) )
def _copy_file_to_output_directory(
self, entry: Entry, source_file_path: str, output_file_name: str
):
"""
Helper function to move a file from the working directory to the output directory.
Will add it to the download archive mapping if the archive is being maintained.
Parameters
----------
entry:
Entry that the file belongs to
source_file_path:
Path to the source file
output_file_name:
Desired output file name with the output directory as its relative directory
"""
destination_file_path = Path(self.output_directory) / Path(output_file_name)
os.makedirs(os.path.dirname(destination_file_path), exist_ok=True)
copyfile(source_file_path, destination_file_path)
if self.maintain_download_archive:
self._enhanced_download_archive.mapping.add_entry(entry, output_file_name)
def _copy_entry_files_to_output_directory(self, entry: Entry): def _copy_entry_files_to_output_directory(self, entry: Entry):
""" """
Helper function to move the media file and optionally thumbnail file to the output directory Helper function to move the media file and optionally thumbnail file to the output directory
@ -181,10 +158,8 @@ class Subscription:
output_file_name = self.overrides.apply_formatter( output_file_name = self.overrides.apply_formatter(
formatter=self.output_options.file_name, entry=entry formatter=self.output_options.file_name, entry=entry
) )
self._copy_file_to_output_directory( self._enhanced_download_archive.save_file(
entry=entry, file_name=entry.get_download_file_path(), output_file_name=output_file_name, entry=entry
source_file_path=entry.get_download_file_path(),
output_file_name=output_file_name,
) )
if self.output_options.thumbnail_name: if self.output_options.thumbnail_name:
@ -195,10 +170,10 @@ class Subscription:
# We always convert entry thumbnails to jpgs, and is performed here # We always convert entry thumbnails to jpgs, and is performed here
convert_download_thumbnail(entry=entry) convert_download_thumbnail(entry=entry)
self._copy_file_to_output_directory( self._enhanced_download_archive.save_file(
entry=entry, file_name=entry.get_download_file_name(),
source_file_path=entry.get_download_thumbnail_path(),
output_file_name=output_thumbnail_name, output_file_name=output_thumbnail_name,
entry=entry,
) )
@contextlib.contextmanager @contextlib.contextmanager
@ -243,30 +218,40 @@ class Subscription:
for plugin_type, plugin_options in self.plugins: for plugin_type, plugin_options in self.plugins:
plugin = plugin_type( plugin = plugin_type(
plugin_options=plugin_options, plugin_options=plugin_options,
output_directory=self.output_directory,
overrides=self.overrides, overrides=self.overrides,
enhanced_download_archive=self._enhanced_download_archive enhanced_download_archive=self._enhanced_download_archive,
if self.maintain_download_archive
else None,
) )
plugins.append(plugin) plugins.append(plugin)
return plugins return plugins
def download(self): def download(self, dry_run: bool = False):
""" """
Performs the subscription download. Performs the subscription download
Parameters
----------
dry_run
If true, do not download any video/audio files or move anything to the output
directory.
""" """
# TODO: Move this logic to separate function
# TODO: set id here as well
ytdl_options = copy.deepcopy(self.ytdl_options.dict)
if dry_run:
ytdl_options["skip_download"] = True
if self.downloader_class.supports_download_archive and self.maintain_download_archive:
ytdl_options["download_archive"] = str(
Path(self.working_directory) / self._enhanced_download_archive.archive_file_name
)
plugins = self._initialize_plugins() plugins = self._initialize_plugins()
with self._prepare_working_directory(), self._maintain_archive_file(): with self._prepare_working_directory(), self._maintain_archive_file():
downloader = self.downloader_class( downloader = self.downloader_class(
working_directory=self.working_directory, working_directory=self.working_directory,
download_options=self.downloader_options, download_options=self.downloader_options,
ytdl_options=self.ytdl_options.dict, ytdl_options=ytdl_options,
download_archive_file_name=self._enhanced_download_archive.archive_file_name
if self.maintain_download_archive
else None,
) )
entries = downloader.download() entries = downloader.download()

View file

@ -0,0 +1,48 @@
import os
from pathlib import Path
from shutil import copyfile
from typing import Set
from typing import Union
class FileHandler:
"""
Performs and tracks all file moving/copying/deleting
"""
def __init__(self, working_directory: str, output_directory: str, dry_run: bool):
self.dry_run = dry_run
self.working_directory = working_directory
self.output_directory = output_directory
self.files_created: Set[str] = set()
self.files_deleted: Set[str] = set()
@classmethod
def copy(cls, src_file_path: Union[str, Path], dst_file_path: Union[str, Path]):
copyfile(src=src_file_path, dst=dst_file_path)
@classmethod
def delete(cls, file_path: Union[str, Path]):
if os.path.isfile(file_path):
os.remove(file_path)
def copy_file_to_output_directory(self, file_name: str, output_file_name: str):
self.files_created.add(output_file_name)
if not self.dry_run:
output_file_path = Path(self.output_directory) / output_file_name
os.makedirs(os.path.dirname(output_file_path), exist_ok=True)
self.copy(
src_file_path=Path(self.working_directory) / file_name,
dst_file_path=output_file_path,
)
def delete_file_from_output_directory(self, file_name: str):
file_path = Path(self.output_directory) / file_name
exists = os.path.isfile(file_path)
if exists:
self.files_deleted.add(file_name)
if not self.dry_run:
self.delete(file_path=file_path)

View file

@ -1,13 +1,14 @@
import contextlib import contextlib
import io import io
import logging import logging
import os
import sys import sys
import tempfile import tempfile
from dataclasses import dataclass from dataclasses import dataclass
from typing import List from typing import List
from typing import Optional from typing import Optional
from ytdl_sub.utils.file_handler import FileHandler
@dataclass @dataclass
class LoggerLevel: class LoggerLevel:
@ -201,5 +202,5 @@ class Logger:
""" """
cls._DEBUG_LOGGER_FILE.close() cls._DEBUG_LOGGER_FILE.close()
if delete_debug_file and os.path.isfile(cls.debug_log_filename()): if delete_debug_file:
os.remove(cls.debug_log_filename()) FileHandler.delete(cls.debug_log_filename())

View file

@ -13,6 +13,7 @@ from typing import Set
from yt_dlp import DateRange from yt_dlp import DateRange
from ytdl_sub.entries.entry import Entry from ytdl_sub.entries.entry import Entry
from ytdl_sub.utils.file_handler import FileHandler
from ytdl_sub.utils.logger import Logger from ytdl_sub.utils.logger import Logger
@ -206,7 +207,7 @@ class DownloadMappings:
entry entry
Entry that this file belongs to Entry that this file belongs to
entry_file_path entry_file_path
Relative path to the file that belongs to the entry Relative path to the file that lives in the output directory
Returns Returns
------- -------
@ -330,13 +331,18 @@ class EnhancedDownloadArchive:
6. ( Delete the working directory ) 6. ( Delete the working directory )
""" """
def __init__(self, subscription_name: str, working_directory: str, output_directory: str): def __init__(
self,
subscription_name: str,
working_directory: str,
output_directory: str,
dry_run: bool = False,
):
self.subscription_name = subscription_name self.subscription_name = subscription_name
self.working_directory = working_directory self._file_handler = FileHandler(
self.output_directory = output_directory working_directory=working_directory, output_directory=output_directory, dry_run=dry_run
)
self._download_archive: Optional[DownloadArchive] = None self._download_mapping = DownloadMappings()
self._download_mapping: Optional[DownloadMappings] = None
self._logger = Logger.get(name=subscription_name) self._logger = Logger.get(name=subscription_name)
@ -349,6 +355,14 @@ class EnhancedDownloadArchive:
""" """
return f".ytdl-subscribe-{self.subscription_name}-download-archive.txt" return f".ytdl-subscribe-{self.subscription_name}-download-archive.txt"
@property
def working_directory(self) -> str:
return self._file_handler.working_directory
@property
def output_directory(self) -> str:
return self._file_handler.output_directory
@property @property
def _mapping_file_name(self) -> str: def _mapping_file_name(self) -> str:
""" """
@ -395,7 +409,6 @@ class EnhancedDownloadArchive:
def _load(self) -> "EnhancedDownloadArchive": def _load(self) -> "EnhancedDownloadArchive":
""" """
Tries to load download mappings if they are present in the output directory. Tries to load download mappings if they are present in the output directory.
If they are not, initialize an empty mapping.
Returns Returns
------- -------
@ -406,14 +419,9 @@ class EnhancedDownloadArchive:
self._download_mapping = DownloadMappings.from_file( self._download_mapping = DownloadMappings.from_file(
json_file_path=self._mapping_output_file_path json_file_path=self._mapping_output_file_path
) )
# Otherwise, init an empty download mappings object. Keep _download_archive as None to
# indicate it was not loaded
else:
self._download_mapping = DownloadMappings()
return self return self
def _copy_to_working_directory(self) -> "EnhancedDownloadArchive": def _copy_mapping_to_working_directory(self) -> "EnhancedDownloadArchive":
""" """
If the mapping is not empty, create a download archive from it and save it into the If the mapping is not empty, create a download archive from it and save it into the
working directory. This will tell YTDL to not redownload already downloaded entries. working directory. This will tell YTDL to not redownload already downloaded entries.
@ -442,7 +450,7 @@ class EnhancedDownloadArchive:
self self
""" """
self._load() self._load()
self._copy_to_working_directory() self._copy_mapping_to_working_directory()
return self return self
def remove_stale_files(self, date_range: DateRange) -> "EnhancedDownloadArchive": def remove_stale_files(self, date_range: DateRange) -> "EnhancedDownloadArchive":
@ -464,12 +472,8 @@ class EnhancedDownloadArchive:
) )
for uid, mapping in stale_mappings.items(): for uid, mapping in stale_mappings.items():
self._logger.info("[%s] Removing the following stale file(s):", uid)
for file_name in mapping.file_names: for file_name in mapping.file_names:
file_path = Path(self.output_directory) / Path(file_name) self._file_handler.delete_file_from_output_directory(file_name=file_name)
self._logger.info(" - %s", file_path)
if os.path.exists(file_path):
os.remove(file_path)
self.mapping.remove_entry(entry_id=uid) self.mapping.remove_entry(entry_id=uid)
@ -486,14 +490,34 @@ class EnhancedDownloadArchive:
# TODO: Make sure this logic is actually right... # TODO: Make sure this logic is actually right...
# Load the download archive from the working directory, which should contain any past # Load the download archive from the working directory, which should contain any past
# and new entries downloaded in this session # and new entries downloaded in this session
self._download_archive = DownloadArchive.from_file(self._archive_working_file_path) download_archive = DownloadArchive.from_file(self._archive_working_file_path)
# Keep the download archive in sync with the mapping # Keep the download archive in sync with the mapping
for entry_id in self.mapping.entry_ids: for entry_id in self.mapping.entry_ids:
if not self._download_archive.contains(entry_id): if not download_archive.contains(entry_id):
self._download_archive.remove_entry(entry_id) download_archive.remove_entry(entry_id)
# Save the updated mapping file to the output directory # Save the updated mapping file to the output directory
self._download_mapping.to_file(output_json_file=self._mapping_output_file_path) self._download_mapping.to_file(output_json_file=self._mapping_output_file_path)
return self return self
def save_file(self, file_name: str, output_file_name: str, entry: Optional[Entry] = None):
"""
Saves a file from the working directory to the output directory
Parameters
----------
file_name
Name of the file to move (does not include working directory path)
output_file_name
Final name of the file in the output directory (does not include output directory path)
entry
Optional. Entry that this file belongs to
"""
if entry:
self.mapping.add_entry(entry=entry, entry_file_path=output_file_name)
self._file_handler.copy_file_to_output_directory(
file_name=file_name, output_file_name=output_file_name
)