This commit is contained in:
Broque Thomas 2025-07-24 17:45:42 -07:00
parent 44c262c7b4
commit 6358de554f
3 changed files with 83 additions and 65 deletions

View file

@ -27,7 +27,7 @@ class MusicMatchingEngine:
# General patterns to remove all content in brackets/parentheses # General patterns to remove all content in brackets/parentheses
r'\(.*\)', r'\(.*\)',
r'\[.*\]', r'\[.*\]',
# General pattern to remove everything after a hyphen # General pattern to remove everything after a hyphen, which is common for version info
r'\s-\s.*', r'\s-\s.*',
# Patterns to remove featuring artists from the title itself # Patterns to remove featuring artists from the title itself
r'\sfeat\.?.*', r'\sfeat\.?.*',
@ -60,8 +60,16 @@ class MusicMatchingEngine:
return text return text
def get_core_string(self, text: str) -> str:
"""Returns a 'core' version of a string with only letters and numbers for a strict comparison."""
if not text:
return ""
# Transliterate, lowercase, and remove everything that isn't a letter or digit.
text = unidecode(text).lower()
return re.sub(r'[^a-z0-9]', '', text)
def clean_title(self, title: str) -> str: def clean_title(self, title: str) -> str:
"""Cleans title by removing common extra info using regex.""" """Cleans title by removing common extra info using regex for fuzzy matching."""
cleaned = title cleaned = title
for pattern in self.title_patterns: for pattern in self.title_patterns:
@ -98,48 +106,43 @@ class MusicMatchingEngine:
return max(0, 1.0 - diff_ratio * 5) return max(0, 1.0 - diff_ratio * 5)
def calculate_match_confidence(self, spotify_track: SpotifyTrack, plex_track: PlexTrackInfo) -> Tuple[float, str]: def calculate_match_confidence(self, spotify_track: SpotifyTrack, plex_track: PlexTrackInfo) -> Tuple[float, str]:
"""Calculates a confidence score for a potential match with a more robust, prioritized logic.""" """Calculates a confidence score using a prioritized model, starting with a strict 'core' title check."""
spotify_title_cleaned = self.clean_title(spotify_track.name) # --- Artist Scoring (calculated once) ---
plex_title_cleaned = self.clean_title(plex_track.title)
# --- Artist Scoring ---
spotify_artists_cleaned = [self.clean_artist(a) for a in spotify_track.artists if a] spotify_artists_cleaned = [self.clean_artist(a) for a in spotify_track.artists if a]
plex_artist_normalized = self.normalize_string(plex_track.artist) plex_artist_normalized = self.normalize_string(plex_track.artist)
plex_artist_cleaned = self.clean_artist(plex_track.artist)
best_artist_score = 0.0 best_artist_score = 0.0
for spotify_artist in spotify_artists_cleaned: for spotify_artist in spotify_artists_cleaned:
if spotify_artist and spotify_artist in plex_artist_normalized: if spotify_artist and spotify_artist in plex_artist_normalized:
best_artist_score = 1.0 best_artist_score = 1.0
break break
score = self.similarity_score(spotify_artist, self.clean_artist(plex_track.artist)) score = self.similarity_score(spotify_artist, plex_artist_cleaned)
if score > best_artist_score: if score > best_artist_score:
best_artist_score = score best_artist_score = score
artist_score = best_artist_score artist_score = best_artist_score
# --- Title and Duration Scoring --- # --- Priority 1: Core Title Match (for exact matches like "Girls", "APT.", "LIL DEMON") ---
spotify_core_title = self.get_core_string(spotify_track.name)
plex_core_title = self.get_core_string(plex_track.title)
if spotify_core_title and spotify_core_title == plex_core_title:
# If the core titles are identical, we are highly confident.
# The final score is a high base (0.9) plus a bonus for artist similarity.
confidence = 0.90 + (artist_score * 0.09) # Max score of 0.99
return confidence, "core_title_match"
# --- Priority 2: Fuzzy Title Match (for variations, typos, etc.) ---
spotify_title_cleaned = self.clean_title(spotify_track.name)
plex_title_cleaned = self.clean_title(plex_track.title)
title_score = self.similarity_score(spotify_title_cleaned, plex_title_cleaned) title_score = self.similarity_score(spotify_title_cleaned, plex_title_cleaned)
duration_score = self.duration_similarity(spotify_track.duration_ms, plex_track.duration if plex_track.duration else 0) duration_score = self.duration_similarity(spotify_track.duration_ms, plex_track.duration if plex_track.duration else 0)
# --- Prioritized Confidence Logic --- # Use a standard weighted calculation if the core titles didn't match
# Priority 1: Near-perfect title and artist match is a very strong signal. confidence = (title_score * 0.60) + (artist_score * 0.30) + (duration_score * 0.10)
if title_score > 0.98 and artist_score > 0.9: match_type = "standard_match"
confidence = 0.98
match_type = "strong_match"
# Priority 2: Exact title match, even with a weaker artist match, should have high confidence.
# This helps with short titles like "Girls" or "LIL DEMON".
elif title_score > 0.98:
confidence = 0.90 + (artist_score * 0.05) # Base of 0.9, with a small artist bonus
match_type = "exact_title_match"
# Priority 3: High title similarity is still a good indicator.
elif title_score > 0.9:
confidence = (title_score * 0.6) + (artist_score * 0.3) + (duration_score * 0.1)
match_type = "high_confidence"
# Default: Standard weighted calculation for all other cases.
else:
confidence = (title_score * 0.5) + (artist_score * 0.3) + (duration_score * 0.2)
match_type = "standard_match"
return confidence, match_type return confidence, match_type

View file

@ -250,43 +250,41 @@ class PlexClient:
def search_tracks(self, title: str, artist: str, limit: int = 15) -> List[PlexTrackInfo]: def search_tracks(self, title: str, artist: str, limit: int = 15) -> List[PlexTrackInfo]:
""" """
Searches for tracks in the Plex music library using a more robust, two-step method Searches for tracks in the Plex music library. If an artist is provided, it
that is more compatible with different Plex server versions. searches within that artist's scope. If the artist is empty or not found,
it falls back to a library-wide search for the title.
""" """
if not self.music_library: if not self.music_library:
logger.warning("Plex music library not found. Cannot perform search.") logger.warning("Plex music library not found. Cannot perform search.")
return [] return []
try: try:
# Step 1: Search for the artist first. This is generally reliable.
artist_results = self.music_library.searchArtists(title=artist, limit=1)
candidate_tracks = [] candidate_tracks = []
if artist_results:
# If artist is found, get all their tracks and filter by title in Python. # If an artist is provided, perform a targeted search.
# This avoids potential API filter issues where special characters in the title if artist:
# might cause the search to fail. artist_results = self.music_library.searchArtists(title=artist, limit=1)
plex_artist = artist_results[0] if artist_results:
all_artist_tracks = plex_artist.tracks() plex_artist = artist_results[0]
all_artist_tracks = plex_artist.tracks()
# Use a case-insensitive substring match to find potential tracks. lower_title = title.lower()
# The matching engine will do the final, more precise comparison later. for track in all_artist_tracks:
lower_title = title.lower() if lower_title in track.title.lower():
for track in all_artist_tracks: candidate_tracks.append(track)
if lower_title in track.title.lower(): else:
candidate_tracks.append(track) # If artist not found, still fall back to a general title search
logger.debug(f"Artist '{artist}' not found. Falling back to title search for '{title}'.")
candidate_tracks = self.music_library.searchTracks(title=title, limit=limit)
# If no artist is provided, perform a library-wide title search directly.
else: else:
# Fallback: If the artist wasn't found, search for the track title logger.debug(f"Performing title-only search for '{title}'.")
# across the entire library. This is less precise but better than nothing.
logger.debug(f"Artist '{artist}' not found. Falling back to title search for '{title}'.")
candidate_tracks = self.music_library.searchTracks(title=title, limit=limit) candidate_tracks = self.music_library.searchTracks(title=title, limit=limit)
# Convert the raw Plex track objects to our simplified PlexTrackInfo dataclass.
# Apply the limit here to the final list of candidates.
tracks = [PlexTrackInfo.from_plex_track(track) for track in candidate_tracks[:limit]] tracks = [PlexTrackInfo.from_plex_track(track) for track in candidate_tracks[:limit]]
if tracks: if tracks:
logger.debug(f"Plex search for title='{title}' by artist='{artist}' found {len(tracks)} potential matches.") logger.debug(f"Plex search for title='{title}', artist='{artist or 'N/A'}' found {len(tracks)} potential matches.")
return tracks return tracks
@ -296,6 +294,7 @@ class PlexClient:
traceback.print_exc() traceback.print_exc()
return [] return []
def get_library_stats(self) -> Dict[str, int]: def get_library_stats(self) -> Dict[str, int]:
if not self.music_library: if not self.music_library:
return {} return {}

View file

@ -120,39 +120,38 @@ class PlaylistTrackAnalysisWorker(QRunnable):
def _check_track_in_plex(self, spotify_track): def _check_track_in_plex(self, spotify_track):
""" """
Check if a Spotify track exists in Plex by trying several search strategies Check if a Spotify track exists in Plex by trying a multi-stage search strategy:
across ALL artists associated with the track. 1. Search using Artist + Title variations.
2. Fallback to searching with only the Title variations across the entire library.
""" """
try: try:
original_title = spotify_track.name original_title = spotify_track.name
# --- Generate a list of title variations --- # --- Generate a list of title variations ---
title_variations = [] title_variations = []
title_variations.append(original_title) # Strategy 1: Original title title_variations.append(original_title)
if " - " in original_title: # Strategy 2: Strip content after hyphen if " - " in original_title:
title_variations.append(original_title.split(' - ')[0].strip()) title_variations.append(original_title.split(' - ')[0].strip())
cleaned_for_search = clean_track_name_for_search(original_title) # Strategy 3: Strip parenthetical content cleaned_for_search = clean_track_name_for_search(original_title)
if cleaned_for_search.lower() != original_title.lower(): if cleaned_for_search.lower() != original_title.lower():
title_variations.append(cleaned_for_search) title_variations.append(cleaned_for_search)
base_title = self.matching_engine.clean_title(original_title) # Strategy 4: Aggressively cleaned title base_title = self.matching_engine.clean_title(original_title)
if base_title.lower() not in [t.lower() for t in title_variations]: if base_title.lower() not in [t.lower() for t in title_variations]:
title_variations.append(base_title) title_variations.append(base_title)
unique_title_variations = list(dict.fromkeys(title_variations)) unique_title_variations = list(dict.fromkeys(title_variations))
# --- Execute searches for EACH artist and collect all potential matches --- # --- Execute searches and collect all potential matches ---
all_potential_matches = [] all_potential_matches = []
found_match_ids = set() found_match_ids = set()
# Use all artists from Spotify, not just the first one # STAGE 1: Search with Artist + Title
artists_to_search = spotify_track.artists if spotify_track.artists else [""] artists_to_search = spotify_track.artists if spotify_track.artists else [""]
for artist_name in artists_to_search: for artist_name in artists_to_search:
if self._cancelled: return None, 0.0 if self._cancelled: return None, 0.0
print(f"🎤 Searching for artist: '{artist_name}'")
for query_title in unique_title_variations: for query_title in unique_title_variations:
if self._cancelled: return None, 0.0 if self._cancelled: return None, 0.0
@ -166,13 +165,28 @@ class PlaylistTrackAnalysisWorker(QRunnable):
if track.id not in found_match_ids: if track.id not in found_match_ids:
all_potential_matches.append(track) all_potential_matches.append(track)
found_match_ids.add(track.id) found_match_ids.add(track.id)
# STAGE 2: Fallback search with Title Only (as suggested)
print(f"🎤 Performing title-only fallback search for '{original_title}'")
for query_title in unique_title_variations:
if self._cancelled: return None, 0.0
# Pass an empty artist to trigger title-only search logic in plex_client
title_only_matches = self.plex_client.search_tracks(
title=query_title,
artist="",
limit=10
)
for track in title_only_matches:
if track.id not in found_match_ids:
all_potential_matches.append(track)
found_match_ids.add(track.id)
if not all_potential_matches: if not all_potential_matches:
print(f"❌ No Plex candidates found for '{original_title}' after trying all artists and title variations.") print(f"❌ No Plex candidates found for '{original_title}' after all search strategies.")
return None, 0.0 return None, 0.0
# --- Use the matching engine to find the best match among ALL candidates --- # --- Use the matching engine to find the best match among ALL candidates ---
print(f"✅ Found {len(all_potential_matches)} potential Plex matches for '{original_title}'. Scoring now...") print(f"✅ Found {len(all_potential_matches)} total potential Plex matches for '{original_title}'. Scoring now...")
match_result = self.matching_engine.find_best_match(spotify_track, all_potential_matches) match_result = self.matching_engine.find_best_match(spotify_track, all_potential_matches)
if match_result.is_match: if match_result.is_match:
@ -188,6 +202,8 @@ class PlaylistTrackAnalysisWorker(QRunnable):
traceback.print_exc() traceback.print_exc()
return None, 0.0 return None, 0.0
class TrackDownloadWorkerSignals(QObject): class TrackDownloadWorkerSignals(QObject):
"""Signals for track download worker""" """Signals for track download worker"""
download_started = pyqtSignal(int, int, str) # download_index, track_index, download_id download_started = pyqtSignal(int, int, str) # download_index, track_index, download_id