Fix MBID Mismatch Detector comparing MB title against DB title (filename) instead of embedded tag

The scan was reading `t.title` from the database for comparison, which is
populated from the filename rather than the file's embedded TITLE tag. Any
track whose filename differs from the clean MB title (e.g. includes artist,
album, bitrate) was incorrectly flagged as a mismatch.

Fix: extend `_read_mbid_from_file` → `_read_file_tags` to also read the
embedded TITLE tag (ID3 TIT2, Vorbis `title`, MP4 `©nam`). The comparison
now uses the embedded title, falling back to the DB title only when no TITLE
tag is present in the file.

Fixes #296
This commit is contained in:
Broque Thomas 2026-04-14 12:19:41 -07:00
parent 751024ec64
commit 0edb8e93bb

View file

@ -54,10 +54,11 @@ def _title_matches(file_title, mb_title):
return ratio >= TITLE_SIMILARITY_THRESHOLD return ratio >= TITLE_SIMILARITY_THRESHOLD
def _read_mbid_from_file(file_path): def _read_file_tags(file_path):
"""Read the MusicBrainz recording MBID from an audio file's tags. """Read the MusicBrainz recording MBID and embedded title from an audio file's tags.
Returns (mbid_string, format_name) or (None, None) if not present. Returns (mbid_string, embedded_title, format_name) or (None, None, None) if not readable.
The embedded_title may be None if no TITLE tag is present.
""" """
try: try:
from mutagen import File as MutagenFile from mutagen import File as MutagenFile
@ -68,42 +69,51 @@ def _read_mbid_from_file(file_path):
audio = MutagenFile(file_path) audio = MutagenFile(file_path)
if audio is None: if audio is None:
return None, None return None, None, None
if isinstance(audio.tags, ID3): if isinstance(audio.tags, ID3):
# MP3: UFID frame # MP3: UFID frame for MBID
mbid = None
ufid_key = f'UFID:{_MBID_TAG_KEYS["mp3_ufid_owner"]}' ufid_key = f'UFID:{_MBID_TAG_KEYS["mp3_ufid_owner"]}'
ufid = audio.tags.get(ufid_key) ufid = audio.tags.get(ufid_key)
if ufid and ufid.data: if ufid and ufid.data:
return ufid.data.decode('ascii', errors='ignore'), 'mp3' mbid = ufid.data.decode('ascii', errors='ignore')
# Also check TXXX fallback (some taggers use this) else:
for key in ['TXXX:MusicBrainz Track Id', 'TXXX:MUSICBRAINZ_TRACKID']: # Also check TXXX fallback (some taggers use this)
txxx = audio.tags.get(key) for key in ['TXXX:MusicBrainz Track Id', 'TXXX:MUSICBRAINZ_TRACKID']:
if txxx and txxx.text: txxx = audio.tags.get(key)
return txxx.text[0], 'mp3' if txxx and txxx.text:
return None, None mbid = txxx.text[0]
break
# Embedded title from TIT2 frame
tit2 = audio.tags.get('TIT2')
embedded_title = tit2.text[0] if tit2 and tit2.text else None
return mbid, embedded_title, 'mp3' if mbid else None
elif isinstance(audio, (FLAC, OggVorbis)): elif isinstance(audio, (FLAC, OggVorbis)):
vals = audio.get(_MBID_TAG_KEYS['vorbis'], []) vals = audio.get(_MBID_TAG_KEYS['vorbis'], [])
if not vals: if not vals:
vals = audio.get('musicbrainz_trackid', []) vals = audio.get('musicbrainz_trackid', [])
if vals: mbid = vals[0] if vals else None
return vals[0], 'flac' if isinstance(audio, FLAC) else 'ogg' title_vals = audio.get('title', [])
return None, None embedded_title = title_vals[0] if title_vals else None
fmt = 'flac' if isinstance(audio, FLAC) else 'ogg'
return mbid, embedded_title, fmt if mbid else None
elif isinstance(audio, MP4): elif isinstance(audio, MP4):
vals = audio.get(_MBID_TAG_KEYS['mp4'], []) vals = audio.get(_MBID_TAG_KEYS['mp4'], [])
mbid = None
if vals: if vals:
raw = vals[0] raw = vals[0]
if isinstance(raw, bytes): mbid = raw.decode('utf-8', errors='ignore') if isinstance(raw, bytes) else str(raw)
return raw.decode('utf-8', errors='ignore'), 'mp4' title_vals = audio.get('\xa9nam', [])
return str(raw), 'mp4' embedded_title = title_vals[0] if title_vals else None
return None, None return mbid, embedded_title, 'mp4' if mbid else None
return None, None return None, None, None
except Exception as e: except Exception as e:
logger.debug("Error reading MBID from %s: %s", file_path, e) logger.debug("Error reading tags from %s: %s", file_path, e)
return None, None return None, None, None
def _remove_mbid_from_file(file_path): def _remove_mbid_from_file(file_path):
@ -275,12 +285,15 @@ class MbidMismatchDetectorJob(RepairJob):
result.scanned += 1 result.scanned += 1
continue continue
# Read MBID from file tags # Read MBID and embedded title from file tags
mbid, fmt = _read_mbid_from_file(resolved) mbid, embedded_title, fmt = _read_file_tags(resolved)
if not mbid: if not mbid:
result.scanned += 1 result.scanned += 1
continue continue
# Use the embedded TITLE tag for comparison; fall back to DB title only if absent
file_title = embedded_title if embedded_title else title
# Validate the MBID against MusicBrainz # Validate the MBID against MusicBrainz
checked += 1 checked += 1
@ -318,13 +331,13 @@ class MbidMismatchDetectorJob(RepairJob):
mb_artist = credit['artist'].get('name', '') mb_artist = credit['artist'].get('name', '')
break break
# Compare: does the MBID's title match the file's title? # Compare: does the MBID's title match the file's embedded title?
if not _title_matches(title, mb_title): if not _title_matches(file_title, mb_title):
self._create_mismatch_finding( self._create_mismatch_finding(
context, result, track_id, title, artist_name, album_title, context, result, track_id, title, artist_name, album_title,
resolved, album_thumb, artist_thumb, mbid, resolved, album_thumb, artist_thumb, mbid,
mb_title=mb_title, mb_artist=mb_artist, mb_title=mb_title, mb_artist=mb_artist,
reason=f'MBID points to "{mb_title}" by {mb_artist}, expected "{title}"' reason=f'MBID points to "{mb_title}" by {mb_artist}, expected "{file_title}"'
) )
except Exception as e: except Exception as e: