watchlist: don't fuse different editions as the same album (Sokhi: Expedition 33)

_normalize_album_for_match stripped ANY trailing '- clause', so a real distinguishing
subtitle ('Clair Obscur: Expedition 33 - Nos vies en Lumière (Bonus Edition)') collapsed to
the same name as the OST → _albums_likely_match treated them as one album → the watchlist
marked unowned tracks of one edition as owned via the other and under-wishlisted.

- strip a trailing '- ...' clause ONLY when every token in it is an edition/format
  qualifier (+ connectors / year-ordinal): '- Single', '- Acoustic Version', '- 2011
  Remaster' still collapse, but real subtitles ('- Nos vies en Lumière', '- Volume 2',
  '- Live in Berlin') are kept. Avoids the inverse regression (a same-album pair splitting
  into a redownload loop), which a naive narrow strip would have caused.
- drop the loose substring shortcut + raise the fuzzy floor 0.6→0.85; genuine drift already
  collapses to an EXACT match, so the looseness only ever produced false fuses.

blast radius: _albums_likely_match has exactly one caller (the allow-duplicates skip).
48 album-match tests pass (qualifier-suffix merges + edition-subtitle splits) + 219 watchlist.
This commit is contained in:
BoulderBadgeDad 2026-06-26 20:30:20 -07:00
parent e96d62432f
commit 7e2d2db08d
2 changed files with 49 additions and 8 deletions

View file

@ -328,6 +328,22 @@ _ALBUM_QUALIFIER_RE = re.compile(
re.IGNORECASE, re.IGNORECASE,
) )
# A trailing "- ..." clause is stripped ONLY when EVERY token in it is an edition/format
# qualifier (+ connectors / a year-ordinal). So "- Single", "- Acoustic Version", "- 2011
# Remaster" collapse to the base name, but a real distinguishing subtitle ("- Nos vies en
# Lumière", "- Live in Berlin") is kept — the bug was a blanket "- anything$" strip that
# erased subtitles and fused different editions (Sokhi: Expedition 33 OST vs Bonus Edition).
_DASH_QUALIFIER_WORD = (
r'live|acoustic|electric|instrumental|unplugged|mono|stereo|demos?|reissue|'
r'remix(?:es)?|edit(?:ed)?|radio|single|ep|lp|version|mix(?:es)?|sessions?|bootleg|'
r'covers?|original|redux|deluxe|expanded|remaster(?:ed)?|anniversary|special|'
r'edition|bonus|extended|explicit|clean|soundtrack|ost|score'
)
_TRAILING_DASH_QUALIFIER_RE = re.compile(
r'\s*-\s*(?:(?:' + _DASH_QUALIFIER_WORD + r'|the|a|and|&|\+|\d+(?:st|nd|rd|th)?)\b[\s\-]*)+$',
re.IGNORECASE,
)
def _normalize_album_for_match(name: str) -> str: def _normalize_album_for_match(name: str) -> str:
"""Return a canonical form of an album name suitable for fuzzy comparison. """Return a canonical form of an album name suitable for fuzzy comparison.
@ -347,8 +363,9 @@ def _normalize_album_for_match(name: str) -> str:
# they're almost always edition or commentary noise, not part of the # they're almost always edition or commentary noise, not part of the
# album's identifying name. # album's identifying name.
cleaned = re.sub(r'\s*[\(\[][^\)\]]*[\)\]]\s*', ' ', cleaned) cleaned = re.sub(r'\s*[\(\[][^\)\]]*[\)\]]\s*', ' ', cleaned)
# Trailing dash-clauses ("Album - Remastered", "Album - Live") # Trailing dash-clause, but ONLY when it's entirely edition/format qualifiers — a real
cleaned = re.sub(r'\s*-\s*[^-]+$', '', cleaned) # subtitle is preserved (see _TRAILING_DASH_QUALIFIER_RE).
cleaned = _TRAILING_DASH_QUALIFIER_RE.sub(' ', cleaned)
cleaned = re.sub(r'[^a-z0-9 ]+', ' ', cleaned.lower()) cleaned = re.sub(r'[^a-z0-9 ]+', ' ', cleaned.lower())
cleaned = re.sub(r'\s+', ' ', cleaned).strip() cleaned = re.sub(r'\s+', ' ', cleaned).strip()
return cleaned return cleaned
@ -382,7 +399,7 @@ def _extract_volume_marker(normalized_name: str):
return last.group(1) or last.group(2) return last.group(1) or last.group(2)
def _albums_likely_match(spotify_album: str, lib_album: str, threshold: float = 0.6) -> bool: def _albums_likely_match(spotify_album: str, lib_album: str, threshold: float = 0.85) -> bool:
"""Return True when two album names plausibly identify the same release. """Return True when two album names plausibly identify the same release.
Designed to swallow naming drift between metadata sources and the Designed to swallow naming drift between metadata sources and the
@ -406,11 +423,11 @@ def _albums_likely_match(spotify_album: str, lib_album: str, threshold: float =
return False return False
if norm_a == norm_b: if norm_a == norm_b:
return True return True
# After normalization the shorter name often becomes a prefix / # No loose substring shortcut: after qualifier-stripping, a short name being a
# substring of the longer one ("napoleon dynamite" ⊂ "napoleon # prefix of a longer one is usually a DIFFERENT edition carrying a real subtitle
# dynamite music from the motion picture" before stripping). # ("clair obscur expedition 33" ⊂ "clair obscur expedition 33 nos vies en lumiere"),
if norm_a in norm_b or norm_b in norm_a: # not naming drift. Genuine drift collapses to an EXACT match above; everything else
return True # must clear a high overall-similarity bar.
return SequenceMatcher(None, norm_a, norm_b).ratio() >= threshold return SequenceMatcher(None, norm_a, norm_b).ratio() >= threshold

View file

@ -156,6 +156,13 @@ def test_compilation_score_explanation() -> None:
("Inception (Music From The Motion Picture)", "Inception Soundtrack"), ("Inception (Music From The Motion Picture)", "Inception Soundtrack"),
# Substring containment # Substring containment
("Random Access Memories", "Random Access Memories (Bonus Edition)"), ("Random Access Memories", "Random Access Memories (Bonus Edition)"),
# Dash-suffixed qualifiers must still collapse — these are the SAME album, so
# treating them as different would re-wishlist/redownload forever (the failure the
# original blanket strip guarded against; the narrowed strip must keep covering it).
("Album Name", "Album Name - Single"),
("Album Name", "Album Name - Acoustic Version"),
("Hotel California", "Hotel California - 2013 Remaster"),
("Album", "Album - The Remixes"),
], ],
) )
def test_likely_match_positive(spotify_name, lib_name) -> None: def test_likely_match_positive(spotify_name, lib_name) -> None:
@ -176,12 +183,29 @@ def test_likely_match_positive(spotify_name, lib_name) -> None:
("Abbey Road", "Sgt. Pepper's Lonely Hearts Club Band"), ("Abbey Road", "Sgt. Pepper's Lonely Hearts Club Band"),
# Same word in title but different album # Same word in title but different album
("Greatest Hits Volume 1", "Greatest Hits Volume 2"), ("Greatest Hits Volume 1", "Greatest Hits Volume 2"),
# Sokhi: distinct editions of the same franchise — the OST vs a bonus edition
# with a real subtitle. USED to collapse to the same normalized name (the blanket
# trailing-dash strip removed "- Nos vies en Lumière"), so the watchlist marked
# unowned OST tracks as owned via the bonus edition. Must be DIFFERENT albums.
("Clair Obscur: Expedition 33: Original Soundtrack",
"Clair Obscur: Expedition 33 - Nos vies en Lumière (Bonus Edition)"),
# A real subtitle after a dash must not be stripped down to the base name.
("The Album", "The Album - A Whole Different Subtitle"),
], ],
) )
def test_likely_match_negative(spotify_name, lib_name) -> None: def test_likely_match_negative(spotify_name, lib_name) -> None:
assert not _albums_likely_match(spotify_name, lib_name) assert not _albums_likely_match(spotify_name, lib_name)
def test_real_subtitle_after_dash_is_preserved() -> None:
# the regression's root cause: a meaningful subtitle must survive normalization,
# while a recognized qualifier after a dash ("- Live", "- 2011") still collapses.
assert _normalize_album_for_match("Clair Obscur: Expedition 33 - Nos vies en Lumière") \
!= _normalize_album_for_match("Clair Obscur: Expedition 33")
assert _normalize_album_for_match("Some Album - Live") == _normalize_album_for_match("Some Album")
assert _normalize_album_for_match("Some Album - 2011") == _normalize_album_for_match("Some Album")
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# _albums_likely_match — defensive cases # _albums_likely_match — defensive cases
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------