From 69ab4d3e46ad6598fec566be3200be2f0322b976 Mon Sep 17 00:00:00 2001 From: Jesse Bannon Date: Mon, 21 Nov 2022 00:05:25 -0800 Subject: [PATCH] [FEATURE] Add support to scrape chapters from description + comments --- src/ytdl_sub/plugins/chapters.py | 11 +++++-- src/ytdl_sub/utils/chapters.py | 45 ++++++++++++++++++++++++-- tests/unit/utils/test_chapters.py | 52 +++++++++++++++++++++++++++++++ 3 files changed, 103 insertions(+), 5 deletions(-) diff --git a/src/ytdl_sub/plugins/chapters.py b/src/ytdl_sub/plugins/chapters.py index 870b144c..1862de82 100644 --- a/src/ytdl_sub/plugins/chapters.py +++ b/src/ytdl_sub/plugins/chapters.py @@ -62,9 +62,6 @@ class ChaptersOptions(PluginOptions): Embeds chapters to video files if they are present. Additional options to add SponsorBlock chapters and remove specific ones. Can also remove chapters using regex. - Note that at this time, chapter removal with regex will not work with chapters added via - timestamp file. - Usage: .. code-block:: yaml @@ -90,6 +87,8 @@ class ChaptersOptions(PluginOptions): _optional_keys = { "embed_chapters", + "allow_chapters_from_description", + "allow_chapters_from_comments", "embed_chapter_timestamps", "sponsorblock_categories", "remove_sponsorblock_categories", @@ -117,6 +116,12 @@ class ChaptersOptions(PluginOptions): self._embed_chapter_timestamps = self._validate_key_if_present( "embed_chapter_timestamps", StringValidator ) + self._allow_chapters_from_description = self._validate_key_if_present( + key="allow_chapters_from_description", validator=BoolValidator, default=False + ).value + self._allow_chapters_from_comments = self._validate_key_if_present( + key="allow_chapters_from_comments", validator=BoolValidator, default=False + ).value if self._remove_sponsorblock_categories and not self._sponsorblock_categories: raise self._validation_exception( diff --git a/src/ytdl_sub/utils/chapters.py b/src/ytdl_sub/utils/chapters.py index 2c016402..da4f26d9 100644 --- a/src/ytdl_sub/utils/chapters.py +++ b/src/ytdl_sub/utils/chapters.py @@ -19,11 +19,11 @@ class Timestamp: # 1:00:00 title # 01:00:00 title # where capture group 1 and 2 are the timestamp and title, respectively - _SPLIT_TIMESTAMP_REGEX = re.compile(r"^((?:\d\d:)?(?:\d:)?(?:\d)?\d:\d\d)$") + TIMESTAMP_REGEX = re.compile(r"((?:\d\d:)?(?:\d:)?(?:\d)?\d:\d\d)") @classmethod def _normalize_timestamp_str(cls, timestamp_str: str) -> str: - match = cls._SPLIT_TIMESTAMP_REGEX.match(timestamp_str) + match = cls.TIMESTAMP_REGEX.match(timestamp_str) if not match: raise ValueError(f"Cannot parse youtube timestamp '{timestamp_str}'") @@ -219,6 +219,44 @@ class Chapters: return cls(timestamps=timestamps, titles=titles) + @classmethod + def from_string(cls, input_str: str) -> "Chapters": + """ + From a string (description or comment), try to extract Chapters. + The scraping logic is simple, if three or more successive lines have timestamps, grab + as many in succession as possible. Remove the timestamp portion to get the chapter title. + + Parameters + ---------- + input_str + String to scrape + + Returns + ------- + Chapters + Could be empty + """ + timestamps: List[Timestamp] = [] + titles: List[str] = [] + + for line in input_str.split("\n"): + # Timestamp captured, store it + if match := Timestamp.TIMESTAMP_REGEX.search(line): + timestamp_str = match.group(1) + timestamps.append(Timestamp.from_str(timestamp_str)) + + # Remove timestamp and surrounding whitespace from it + title_str = re.sub(f"\\s*{re.escape(timestamp_str)}\\s*", " ", line).strip() + titles.append(title_str) + elif len(timestamps) >= 3: + return Chapters(timestamps=timestamps, titles=titles) + # Timestamp was not stored, if only contained 1, reset + else: + timestamps = [] + titles = [] + + return Chapters(timestamps=timestamps, titles=titles) + @classmethod def from_embedded_chapters(cls, file_path: str) -> "Chapters": """ @@ -280,3 +318,6 @@ class Chapters: titles.append(chapter["title"]) return Chapters(timestamps=timestamps, titles=titles) + + def __len__(self): + return len(self.timestamps) diff --git a/tests/unit/utils/test_chapters.py b/tests/unit/utils/test_chapters.py index 931fefc9..aefab07f 100644 --- a/tests/unit/utils/test_chapters.py +++ b/tests/unit/utils/test_chapters.py @@ -1,5 +1,6 @@ import pytest +from ytdl_sub.utils.chapters import Chapters from ytdl_sub.utils.chapters import Timestamp @@ -20,3 +21,54 @@ class TestTimestamp: def test_timestamp_from_str(self, timestamp_str, timestamp_int): ts = Timestamp.from_str(timestamp_str=timestamp_str) assert ts.timestamp_sec == timestamp_int + + +@pytest.fixture +def chapter_description_1() -> str: + return """Support the artist by purchasing the record here: +https://levanika.bandcamp.com/album/p... + +Album Art by: po + +Tracklist: +00:00 intro +00:58 audioclip p94568 +01:18 saskipao +03:32 fifqeby +05:40 dream +07:22 rulji +08:54 modymody +11:23 .ishos +16:09 pos +18:47 lamazybalaxy + +Denivarlevy Socials: + +https://open.spotify.com/artist/2XdIT... +https://youtube.com/channel/UCgI_vAC3... +""" + + +@pytest.fixture +def chapter_description_2() -> str: + return """01. 00:00 Ocean +02. 02:41 Dreams +03. 05:16 Future Tales +04. 08:50 Mind Travelling +05. 11:05 Love Supreme +06. 14:17 Reflections +07. 16:32 Moonlight Fading +08. 19:40 Between Two Worlds +""" + + +class TestChapters: + def test_chapters_from_str_1(self, chapter_description_1): + chapters = Chapters.from_string(chapter_description_1) + assert len(chapters) == 10 + assert chapters.timestamps[-1].readable_str == "18:47" + + def test_chapters_from_str_2(self, chapter_description_2): + chapters = Chapters.from_string(chapter_description_2) + assert len(chapters) == 8 + assert chapters.timestamps[1].readable_str == "2:41" \ No newline at end of file