pinchflat/lib/pinchflat/fast_indexing/youtube_rss.ex
Kieran 33baa99aae Refactor modules into contexts (#78)
* [WIP] break out a few contexts, start refactoring fast index modules

* [WIP] more contexts, this time around slow indexing and downloads

* [WIP] got all tests passing

* [WIP] Added moduledocs

* Built a genserver to rename old jobs on boot

* Added a module naming check; moved things around

* Fixed specs
2024-03-12 17:54:55 -07:00

51 lines
1.5 KiB
Elixir

defmodule Pinchflat.FastIndexing.YoutubeRss do
@moduledoc """
Methods for interacting with YouTube RSS feeds
"""
require Logger
alias Pinchflat.Sources.Source
@doc """
Fetches the recent media IDs from a YouTube RSS feed for a given source.
Returns {:ok, [binary()]} | {:error, binary()}
"""
def get_recent_media_ids_from_rss(%Source{} = source) do
Logger.debug("Fetching recent media IDs from YouTube RSS feed for source: #{source.collection_id}")
case http_client().get(rss_url_for_source(source)) do
{:ok, response} ->
response = to_string(response)
media_id_regex = ~r/<yt:videoId>(.*?)<\/yt:videoId>/
# Don't get on me about using regex to search XML.
# The content is known, well-formed, and simple.
media_ids =
media_id_regex
|> Regex.scan(response)
|> Enum.map(fn [_, id] -> String.trim(id) end)
|> Enum.filter(&(String.length(&1) > 0))
|> Enum.uniq()
Logger.debug("Media ids fetched from RSS: #{inspect(media_ids)}")
{:ok, media_ids}
{:error, _reason} ->
{:error, "Failed to fetch RSS feed"}
end
end
defp rss_url_for_source(source) do
case source.collection_type do
:channel -> "https://www.youtube.com/feeds/videos.xml?channel_id=#{source.collection_id}"
:playlist -> "https://www.youtube.com/feeds/videos.xml?playlist_id=#{source.collection_id}"
end
end
defp http_client do
Application.get_env(:pinchflat, :http_client, Pinchflat.HTTP.HTTPClient)
end
end