Normalize metadata payloads around source

- remove legacy _source/provider emission from the new metadata path
- update import, discovery, wishlist, and repair consumers to read source
- refresh the affected tests to match the new contract
This commit is contained in:
Antti Kettunen 2026-05-05 08:04:41 +03:00
parent 896b1b1012
commit c5c085fb50
No known key found for this signature in database
GPG key ID: C6B2A3D250359BD7
16 changed files with 30 additions and 61 deletions

View file

@ -184,7 +184,7 @@ def run_playlist_discovery_worker(playlists, automation_id=None, deps: PlaylistD
if cached_match and deps.validate_discovery_cache_artist(artist_name, cached_match): if cached_match and deps.validate_discovery_cache_artist(artist_name, cached_match):
extra_data = { extra_data = {
'discovered': True, 'discovered': True,
'provider': discovery_source, 'source': discovery_source,
'confidence': cached_match.get('confidence', 0.85), 'confidence': cached_match.get('confidence', 0.85),
'matched_data': cached_match, 'matched_data': cached_match,
} }
@ -309,7 +309,7 @@ def run_playlist_discovery_worker(playlists, automation_id=None, deps: PlaylistD
extra_data = { extra_data = {
'discovered': True, 'discovered': True,
'provider': discovery_source, 'source': discovery_source,
'confidence': best_confidence, 'confidence': best_confidence,
'matched_data': matched_data, 'matched_data': matched_data,
} }
@ -337,7 +337,7 @@ def run_playlist_discovery_worker(playlists, automation_id=None, deps: PlaylistD
stub = deps.build_discovery_wing_it_stub(track_name, artist_name, duration_ms) stub = deps.build_discovery_wing_it_stub(track_name, artist_name, duration_ms)
extra_data = { extra_data = {
'discovered': True, 'discovered': True,
'provider': 'wing_it_fallback', 'source': 'wing_it_fallback',
'confidence': 0, 'confidence': 0,
'wing_it_fallback': True, 'wing_it_fallback': True,
'matched_data': stub, 'matched_data': stub,

View file

@ -243,7 +243,6 @@ def _normalize_track_match(track_item: Any, provider: str) -> dict:
'preview_url': _extract_lookup_value(track_item, 'preview_url', default=None), 'preview_url': _extract_lookup_value(track_item, 'preview_url', default=None),
'external_urls': _extract_lookup_value(track_item, 'external_urls', default={}) or {}, 'external_urls': _extract_lookup_value(track_item, 'external_urls', default={}) or {},
'popularity': _extract_lookup_value(track_item, 'popularity', default=0) or 0, 'popularity': _extract_lookup_value(track_item, 'popularity', default=0) or 0,
'provider': provider,
'source': provider, 'source': provider,
} }
if not track_data['image_url']: if not track_data['image_url']:
@ -612,7 +611,7 @@ def run_quality_scanner(scope='watchlist', profile_id=1, deps: QualityScannerDep
'bitrate': bitrate, 'bitrate': bitrate,
'matched': matched, 'matched': matched,
'match_id': matched_track_data['id'] if matched_track_data else None, 'match_id': matched_track_data['id'] if matched_track_data else None,
'provider': best_source if matched else None, 'source': best_source if matched else None,
'spotify_id': matched_track_data['id'] if matched_track_data else None, 'spotify_id': matched_track_data['id'] if matched_track_data else None,
} }

View file

@ -349,7 +349,7 @@ def run_youtube_discovery_worker(url_hash, deps: YoutubeDiscoveryDeps):
if result.get('status_class') in ('found', 'wing-it') and result.get('matched_data'): if result.get('status_class') in ('found', 'wing-it') and result.get('matched_data'):
extra_data = { extra_data = {
'discovered': True, 'discovered': True,
'provider': result.get('discovery_source', discovery_source), 'source': result.get('discovery_source', discovery_source),
'confidence': result.get('confidence', 0), 'confidence': result.get('confidence', 0),
'matched_data': result['matched_data'], 'matched_data': result['matched_data'],
} }
@ -357,13 +357,13 @@ def run_youtube_discovery_worker(url_hash, deps: YoutubeDiscoveryDeps):
extra_data['manual_match'] = True extra_data['manual_match'] = True
if result.get('wing_it_fallback'): if result.get('wing_it_fallback'):
extra_data['wing_it_fallback'] = True extra_data['wing_it_fallback'] = True
extra_data['provider'] = 'wing_it_fallback' extra_data['source'] = 'wing_it_fallback'
db.update_mirrored_track_extra_data(db_track_id, extra_data) db.update_mirrored_track_extra_data(db_track_id, extra_data)
else: else:
extra_data = { extra_data = {
'discovered': False, 'discovered': False,
'discovery_attempted': True, 'discovery_attempted': True,
'provider': discovery_source, 'source': discovery_source,
} }
db.update_mirrored_track_extra_data(db_track_id, extra_data) db.update_mirrored_track_extra_data(db_track_id, extra_data)
logger.info(f"Wrote discovery results to DB for {url_hash}") logger.info(f"Wrote discovery results to DB for {url_hash}")

View file

@ -96,16 +96,6 @@ def _normalize_album_source(album: Dict[str, Any], source: str = "") -> str:
return str(album_source).strip().lower() return str(album_source).strip().lower()
def _strip_legacy_source_fields(payload: Any) -> Any:
if not isinstance(payload, dict):
return payload
cleaned = dict(payload)
cleaned.pop("_source", None)
cleaned.pop("provider", None)
return cleaned
def _extract_track_artist_name(track: Dict[str, Any]) -> str: def _extract_track_artist_name(track: Dict[str, Any]) -> str:
artists = track.get("artists") or [] artists = track.get("artists") or []
if isinstance(artists, (str, bytes)): if isinstance(artists, (str, bytes)):
@ -138,8 +128,6 @@ def _coerce_track_int(value: Any, default: int = 1) -> int:
def _normalize_match_track(track: Dict[str, Any], source: str, album: Dict[str, Any]) -> Dict[str, Any]: def _normalize_match_track(track: Dict[str, Any], source: str, album: Dict[str, Any]) -> Dict[str, Any]:
track_album = track.get("album") if isinstance(track.get("album"), dict) else album track_album = track.get("album") if isinstance(track.get("album"), dict) else album
if isinstance(track_album, dict):
track_album = _strip_legacy_source_fields(track_album)
track_source = _normalize_album_source(track, source) track_source = _normalize_album_source(track, source)
track_artists = _normalize_artist_entries(track.get("artists") or []) track_artists = _normalize_artist_entries(track.get("artists") or [])
@ -284,7 +272,6 @@ def build_album_import_context(
else: else:
artist_ctx = resolve_album_artist_context(album, source) artist_ctx = resolve_album_artist_context(album, source)
artist_ctx = _strip_legacy_source_fields(artist_ctx)
artist_ctx.setdefault("genres", []) artist_ctx.setdefault("genres", [])
artist_ctx.setdefault("source", source) artist_ctx.setdefault("source", source)
artist_ctx["genres"] = artist_ctx.get("genres") or [] artist_ctx["genres"] = artist_ctx.get("genres") or []
@ -382,10 +369,6 @@ def build_album_import_context(
} }
normalized_context = normalize_import_context(context) normalized_context = normalize_import_context(context)
normalized_context["artist"] = _strip_legacy_source_fields(normalized_context.get("artist"))
normalized_context["album"] = _strip_legacy_source_fields(normalized_context.get("album"))
normalized_context["track_info"] = _strip_legacy_source_fields(normalized_context.get("track_info"))
normalized_context["original_search_result"] = _strip_legacy_source_fields(normalized_context.get("original_search_result"))
return normalized_context return normalized_context
@ -405,7 +388,7 @@ def build_album_import_match_payload(
source=source, source=source,
) )
album = _strip_legacy_source_fields(dict(album_response.get("album") or {})) album = dict(album_response.get("album") or {})
source = _normalize_album_source(album, album_response.get("source") or source or "") source = _normalize_album_source(album, album_response.get("source") or source or "")
tracks = list(album_response.get("tracks") or []) tracks = list(album_response.get("tracks") or [])
if not album_response.get("success") or not tracks: if not album_response.get("success") or not tracks:

View file

@ -55,11 +55,11 @@ def extract_artist_name(artist: Any) -> str:
def normalize_import_context(context: Optional[Dict[str, Any]]) -> Dict[str, Any]: def normalize_import_context(context: Optional[Dict[str, Any]]) -> Dict[str, Any]:
"""Normalize an import context to neutral fields in place and drop legacy aliases.""" """Normalize an import context to neutral fields in place."""
if not isinstance(context, dict): if not isinstance(context, dict):
return {} return {}
source = context.get("source") or context.get("_source") or "" source = context.get("source") or ""
artist = _as_dict(context.get("artist") or context.get("spotify_artist")) artist = _as_dict(context.get("artist") or context.get("spotify_artist"))
album = _as_dict(context.get("album") or context.get("spotify_album")) album = _as_dict(context.get("album") or context.get("spotify_album"))
track_info = _as_dict(context.get("track_info")) track_info = _as_dict(context.get("track_info"))
@ -73,7 +73,6 @@ def normalize_import_context(context: Optional[Dict[str, Any]]) -> Dict[str, Any
context["album"] = album context["album"] = album
context["track_info"] = track_info context["track_info"] = track_info
context["original_search_result"] = normalized_search context["original_search_result"] = normalized_search
context.pop("_source", None)
context.pop("spotify_artist", None) context.pop("spotify_artist", None)
context.pop("spotify_album", None) context.pop("spotify_album", None)

View file

@ -251,7 +251,7 @@ def _build_single_import_context_payload(
'album_id': album_id, 'album_id': album_id,
'album_type': album_type, 'album_type': album_type,
'release_date': release_date, 'release_date': release_date,
'_source': source or '', 'source': source or '',
} }
album_payload = { album_payload = {
@ -263,14 +263,14 @@ def _build_single_import_context_payload(
'image_url': album_image_url, 'image_url': album_image_url,
'images': album_images, 'images': album_images,
'artists': album_artists, 'artists': album_artists,
'_source': source or '', 'source': source or '',
} }
artist_payload = { artist_payload = {
'id': primary_artist_id, 'id': primary_artist_id,
'name': primary_artist_name, 'name': primary_artist_name,
'genres': [], 'genres': [],
'_source': source or '', 'source': source or '',
} }
original_search = { original_search = {
@ -285,7 +285,7 @@ def _build_single_import_context_payload(
'artists': track_info['artists'], 'artists': track_info['artists'],
'duration_ms': track_info['duration_ms'], 'duration_ms': track_info['duration_ms'],
'id': track_id, 'id': track_id,
'_source': source or '', 'source': source or '',
} }
return { return {
@ -322,7 +322,7 @@ def _build_single_import_fallback_context(
'id': '', 'id': '',
'name': artist_name, 'name': artist_name,
'genres': [], 'genres': [],
'_source': '', 'source': '',
}, },
'album': { 'album': {
'id': '', 'id': '',
@ -333,7 +333,7 @@ def _build_single_import_fallback_context(
'image_url': '', 'image_url': '',
'images': [], 'images': [],
'artists': [], 'artists': [],
'_source': '', 'source': '',
}, },
'track_info': { 'track_info': {
'id': '', 'id': '',
@ -347,7 +347,7 @@ def _build_single_import_fallback_context(
'album_id': '', 'album_id': '',
'album_type': 'album', 'album_type': 'album',
'release_date': '', 'release_date': '',
'_source': '', 'source': '',
}, },
'original_search_result': { 'original_search_result': {
'title': title, 'title': title,
@ -361,7 +361,7 @@ def _build_single_import_fallback_context(
'artists': [{'name': artist_name}], 'artists': [{'name': artist_name}],
'duration_ms': 0, 'duration_ms': 0,
'id': '', 'id': '',
'_source': '', 'source': '',
}, },
'is_album_download': False, 'is_album_download': False,
'has_clean_metadata': False, 'has_clean_metadata': False,

View file

@ -95,7 +95,7 @@ def redownload_start(track_id):
# Fetch full track details from the metadata source for pipeline parity # Fetch full track details from the metadata source for pipeline parity
# This gives us track_number, disc_number, full album data # This gives us track_number, disc_number, full album data
meta_source = metadata.get('_source', '') meta_source = metadata.get('source', '')
meta_id = metadata.get('id', '') meta_id = metadata.get('id', '')
full_track_details = None full_track_details = None
full_album_data = None full_album_data = None

View file

@ -342,8 +342,6 @@ def _build_album_track_entry(track_item: Any, album_info: Dict[str, Any], source
'uri': _extract_lookup_value(track_item, 'uri', default='') or '', 'uri': _extract_lookup_value(track_item, 'uri', default='') or '',
'album': album_info, 'album': album_info,
'source': source, 'source': source,
'provider': source,
'_source': source,
} }
@ -360,8 +358,6 @@ def _build_album_tracks_payload(
album_name=album_name, artist_name=artist_name, source=source, album_name=album_name, artist_name=artist_name, source=source,
) )
album_info['source'] = source album_info['source'] = source
album_info['_source'] = source
album_info['provider'] = source
track_items = _extract_album_track_items(album_data, tracks_data) track_items = _extract_album_track_items(album_data, tracks_data)
tracks = [_build_album_track_entry(track, album_info, source) for track in track_items] tracks = [_build_album_track_entry(track, album_info, source) for track in track_items]

View file

@ -105,8 +105,8 @@ def playlist_explorer_build_tree(deps: PlaylistExplorerDeps):
# Only use discovery data if it matches the active metadata source # Only use discovery data if it matches the active metadata source
is_discovered = extra.get('discovered', False) is_discovered = extra.get('discovered', False)
provider = (extra.get('provider') or '').lower() source = (extra.get('source') or '').lower()
source_matches = provider == source_name or (provider in ('itunes', 'apple') and source_name == 'itunes') source_matches = source == source_name or (source in ('itunes', 'apple') and source_name == 'itunes')
matched = extra.get('matched_data', {}) if (is_discovered and source_matches) else {} matched = extra.get('matched_data', {}) if (is_discovered and source_matches) else {}
artists_list = matched.get('artists', []) artists_list = matched.get('artists', [])
@ -359,5 +359,3 @@ def playlist_explorer_build_tree(deps: PlaylistExplorerDeps):
import traceback import traceback
traceback.print_exc() traceback.print_exc()
return deps.flask_jsonify({"success": False, "error": str(e)}), 500 return deps.flask_jsonify({"success": False, "error": str(e)}), 500

View file

@ -365,7 +365,7 @@ class AlbumCompletenessJob(RepairJob):
'track_number': tn, 'track_number': tn,
'name': item.get('name', ''), 'name': item.get('name', ''),
'disc_number': item.get('disc_number', 1), 'disc_number': item.get('disc_number', 1),
'source': item.get('_source', primary_source), 'source': item.get('source', primary_source),
'source_track_id': item.get('id', ''), 'source_track_id': item.get('id', ''),
'track_id': item.get('id', ''), 'track_id': item.get('id', ''),
'spotify_track_id': item.get('id', ''), 'spotify_track_id': item.get('id', ''),

View file

@ -193,11 +193,7 @@ class WishlistService:
"track_name": track_name, "track_name": track_name,
"artist_name": artist_name, "artist_name": artist_name,
"album_name": album_name, "album_name": album_name,
"provider": ( "source": track_data.get("source") if isinstance(track_data, dict) else None,
track_data.get("provider") or track_data.get("source")
if isinstance(track_data, dict)
else None
),
"spotify_track_id": wishlist_track["spotify_track_id"], "spotify_track_id": wishlist_track["spotify_track_id"],
"spotify_data": track_data, "spotify_data": track_data,
"failure_reason": wishlist_track["failure_reason"], "failure_reason": wishlist_track["failure_reason"],

View file

@ -270,7 +270,7 @@ def test_match_above_threshold_writes_extra_data():
assert len(deps._db.extra_data_writes) == 1 assert len(deps._db.extra_data_writes) == 1
_, extra = deps._db.extra_data_writes[0] _, extra = deps._db.extra_data_writes[0]
assert extra['discovered'] is True assert extra['discovered'] is True
assert extra['provider'] == 'spotify' assert extra['source'] == 'spotify'
assert extra['confidence'] == 0.92 assert extra['confidence'] == 0.92
assert deps._db.cache_saves # saved to cache assert deps._db.cache_saves # saved to cache
@ -289,7 +289,7 @@ def test_match_below_threshold_falls_back_to_wing_it():
assert len(deps._db.extra_data_writes) == 1 assert len(deps._db.extra_data_writes) == 1
_, extra = deps._db.extra_data_writes[0] _, extra = deps._db.extra_data_writes[0]
assert extra['provider'] == 'wing_it_fallback' assert extra['source'] == 'wing_it_fallback'
assert extra['wing_it_fallback'] is True assert extra['wing_it_fallback'] is True

View file

@ -326,7 +326,6 @@ def test_low_quality_tracks_follow_source_priority(mock_db_and_wishlist):
assert spotify_client.search_calls == [] assert spotify_client.search_calls == []
assert len(ws.added) == 1 assert len(ws.added) == 1
add_args = ws.added[0] add_args = ws.added[0]
assert add_args['track_data']['provider'] == 'deezer'
assert add_args['track_data']['source'] == 'deezer' assert add_args['track_data']['source'] == 'deezer'

View file

@ -64,9 +64,9 @@ def test_normalize_import_context_promotes_neutral_fields_without_legacy_aliases
assert get_import_has_full_metadata(normalized) is False assert get_import_has_full_metadata(normalized) is False
def test_normalize_import_context_promotes_legacy_source_alias(): def test_normalize_import_context_keeps_source_field():
context = { context = {
"_source": "spotify", "source": "spotify",
"artist": {"name": "Artist One", "id": "artist-1"}, "artist": {"name": "Artist One", "id": "artist-1"},
"album": {"name": "Album One", "id": "album-1"}, "album": {"name": "Album One", "id": "album-1"},
"track_info": {"name": "Song One", "id": "track-1"}, "track_info": {"name": "Song One", "id": "track-1"},
@ -76,7 +76,6 @@ def test_normalize_import_context_promotes_legacy_source_alias():
normalized = normalize_import_context(context) normalized = normalize_import_context(context)
assert normalized["source"] == "spotify" assert normalized["source"] == "spotify"
assert "_source" not in normalized
assert get_import_source(normalized) == "spotify" assert get_import_source(normalized) == "spotify"

View file

@ -269,7 +269,7 @@ def test_discovered_artist_grouping_uses_matched_data():
'album_name': 'Local Album', 'album_name': 'Local Album',
'extra_data': json.dumps({ 'extra_data': json.dumps({
'discovered': True, 'discovered': True,
'provider': 'spotify', 'source': 'spotify',
'matched_data': { 'matched_data': {
'artists': [{'name': 'Discovered Artist', 'id': 'sp-aid'}], 'artists': [{'name': 'Discovered Artist', 'id': 'sp-aid'}],
'album': {'name': 'Discovered Album'}, 'album': {'name': 'Discovered Album'},
@ -306,7 +306,7 @@ def test_provider_mismatch_falls_back_to_raw_track_name():
'album_name': 'Raw Album', 'album_name': 'Raw Album',
'extra_data': json.dumps({ 'extra_data': json.dumps({
'discovered': True, 'discovered': True,
'provider': 'itunes', # mismatch 'source': 'itunes', # mismatch
'matched_data': { 'matched_data': {
'artists': [{'name': 'iTunes Artist'}], 'artists': [{'name': 'iTunes Artist'}],
}, },

View file

@ -163,7 +163,7 @@ def test_get_wishlist_tracks_for_download_formats_modal_shape():
"track_name": "Song One", "track_name": "Song One",
"artist_name": "Artist One", "artist_name": "Artist One",
"album_name": "Album One", "album_name": "Album One",
"provider": None, "source": None,
"spotify_data": { "spotify_data": {
"id": "sp-1", "id": "sp-1",
"name": "Song One", "name": "Song One",