From bf0bfe59766ccf7501f0551422d8888f83c8f16f Mon Sep 17 00:00:00 2001 From: Antti Kettunen Date: Tue, 5 May 2026 08:04:50 +0300 Subject: [PATCH] Fix Deezer and iTunes normalization edge cases - derive Deezer track album types from album metadata instead of the raw track type - inject Deezer artist data into artist-album payloads - keep the iTunes explicit-duplicate guard from swapping in broken releases --- core/metadata/normalize.py | 6 ++++-- core/metadata/providers/deezer.py | 20 +++++++++++++++++++- core/metadata/providers/itunes.py | 17 +++++++++++++++++ 3 files changed, 40 insertions(+), 3 deletions(-) diff --git a/core/metadata/normalize.py b/core/metadata/normalize.py index 4cb670c5..222a1678 100644 --- a/core/metadata/normalize.py +++ b/core/metadata/normalize.py @@ -230,10 +230,12 @@ def normalize_deezer_track(raw: dict[str, Any]) -> MetadataTrack: contributors = raw.get("contributors") or [] artists = _normalize_artists(contributors if len(_as_list(contributors)) > 1 else [artist] if artist else []) image_url = _first_non_empty(album.get("cover_xl"), album.get("cover_big"), album.get("cover_medium")) - album_type = raw.get("type") or album.get("type") - if not album_type: + album_type = str(album.get("type") or "").lower() + if not album_type or album_type == "track": nb_tracks = int(album.get("nb_tracks", 0) or 0) album_type = _infer_album_type_from_track_count(nb_tracks) + elif album_type == "compile": + album_type = "compilation" return MetadataTrack( id=str(raw.get("id", "")), name=str(raw.get("title", "")), diff --git a/core/metadata/providers/deezer.py b/core/metadata/providers/deezer.py index 4b953e29..3abb0e71 100644 --- a/core/metadata/providers/deezer.py +++ b/core/metadata/providers/deezer.py @@ -86,12 +86,30 @@ class DeezerMetadataAdapter(BaseMetadataAdapter): albums: list[dict[str, Any]] = [] offset = 0 page_size = 100 + artist_data = self.get_artist_raw(artist_id) or {} + artist_name = str(artist_data.get("name", "") or "").strip() + artist_stub: dict[str, Any] = {"id": artist_data.get("id") or artist_id} + if artist_name: + artist_stub["name"] = artist_name while offset < limit: fetch_limit = min(page_size, limit - offset) data = self._request_api(f"artist/{artist_id}/albums", {"limit": fetch_limit, "index": offset}) if not data or not data.get("data"): break - albums.extend(list(data.get("data") or [])) + for item in list(data.get("data") or []): + if not isinstance(item, dict): + continue + enriched = dict(item) + album_artist = enriched.get("artist") + if isinstance(album_artist, dict): + merged_artist = dict(artist_stub) + merged_artist.update(album_artist) + enriched["artist"] = merged_artist + else: + enriched["artist"] = dict(artist_stub) + if artist_name and not enriched.get("artist_name"): + enriched["artist_name"] = artist_name + albums.append(enriched) if len(data.get("data") or []) < fetch_limit: break offset += len(data.get("data") or []) diff --git a/core/metadata/providers/itunes.py b/core/metadata/providers/itunes.py index cbb1501d..3a683334 100644 --- a/core/metadata/providers/itunes.py +++ b/core/metadata/providers/itunes.py @@ -157,6 +157,7 @@ class ITunesMetadataAdapter(BaseMetadataAdapter): return [] seen: dict[str, dict[str, Any]] = {} + track_count_cache: dict[str, int] = {} def _normalize_album_name(name: str) -> str: normalized = (name or "").lower().strip() @@ -170,12 +171,28 @@ class ITunesMetadataAdapter(BaseMetadataAdapter): normalized = re.sub(r"\s+", " ", normalized).strip() return normalized + def _album_has_tracks(collection_id: str) -> bool: + if not collection_id: + return False + if collection_id not in track_count_cache: + items = self._lookup_raw(id=collection_id, entity="song", limit=200) + track_count_cache[collection_id] = sum( + 1 + for item in items + if item.get("wrapperType") == "track" and item.get("kind") == "song" + ) + return track_count_cache[collection_id] > 0 + for album_data in results: if album_data.get("wrapperType") != "collection": continue normalized_name = _normalize_album_name(str(album_data.get("collectionName", "") or "")) + collection_id = str(album_data.get("collectionId", "") or "") current = seen.get(normalized_name) is_explicit = album_data.get("collectionExplicitness") == "explicit" + has_tracks = not is_explicit or _album_has_tracks(collection_id) + if not has_tracks: + continue if current is None: seen[normalized_name] = {"data": album_data, "is_explicit": is_explicit} continue