Fix Picard albumartist orphan false positives
Teach the orphan file detector to match tracked files by both track artist and album artist. This prevents Picard-style albumartist/album (year)/track layouts from being reported as orphans when the DB track artist differs from the album artist. Also check file albumartist tags during fallback matching and add a regression test for the reported Picard folder layout.
This commit is contained in:
parent
4179926899
commit
5d1f3c1b48
2 changed files with 116 additions and 12 deletions
|
|
@ -67,22 +67,28 @@ class OrphanFileDetectorJob(RepairJob):
|
||||||
suffix = '/'.join(parts[-depth:]).lower()
|
suffix = '/'.join(parts[-depth:]).lower()
|
||||||
known_suffixes.add(suffix)
|
known_suffixes.add(suffix)
|
||||||
|
|
||||||
# Build title+artist sets for tag-based fallback matching
|
# Build title+artist sets for fallback matching. Include both
|
||||||
|
# track artist and album artist so Picard-style albumartist paths
|
||||||
|
# don't look orphaned when tracks have featured/guest artists.
|
||||||
cursor.execute("""
|
cursor.execute("""
|
||||||
SELECT t.title, ar.name FROM tracks t
|
SELECT t.title, track_ar.name, album_ar.name FROM tracks t
|
||||||
LEFT JOIN artists ar ON ar.id = t.artist_id
|
LEFT JOIN artists track_ar ON track_ar.id = t.artist_id
|
||||||
|
LEFT JOIN albums al ON al.id = t.album_id
|
||||||
|
LEFT JOIN artists album_ar ON album_ar.id = al.artist_id
|
||||||
WHERE t.title IS NOT NULL AND t.title != ''
|
WHERE t.title IS NOT NULL AND t.title != ''
|
||||||
""")
|
""")
|
||||||
for row in cursor.fetchall():
|
for row in cursor.fetchall():
|
||||||
title = (row[0] or '').lower().strip()
|
title = (row[0] or '').lower().strip()
|
||||||
artist = (row[1] or '').lower().strip()
|
|
||||||
if title:
|
if title:
|
||||||
known_titles.add((title, artist))
|
for artist_value in (row[1], row[2]):
|
||||||
# Also store normalized version (stripped of feat., parentheticals, etc.)
|
artist = (artist_value or '').lower().strip()
|
||||||
clean_t = _strip_extras(title)
|
known_titles.add((title, artist))
|
||||||
clean_a = _strip_extras(artist)
|
# Also store normalized version (stripped of feat.,
|
||||||
if clean_t:
|
# parentheticals, etc.)
|
||||||
known_titles_clean.add((clean_t, clean_a))
|
clean_t = _strip_extras(title)
|
||||||
|
clean_a = _strip_extras(artist)
|
||||||
|
if clean_t:
|
||||||
|
known_titles_clean.add((clean_t, clean_a))
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error("Error reading known file paths from DB: %s", e, exc_info=True)
|
logger.error("Error reading known file paths from DB: %s", e, exc_info=True)
|
||||||
result.errors += 1
|
result.errors += 1
|
||||||
|
|
@ -144,14 +150,21 @@ class OrphanFileDetectorJob(RepairJob):
|
||||||
if audio:
|
if audio:
|
||||||
file_title = ((audio.get('title') or [None])[0] or '').lower().strip()
|
file_title = ((audio.get('title') or [None])[0] or '').lower().strip()
|
||||||
file_artist = ((audio.get('artist') or [None])[0] or '').lower().strip()
|
file_artist = ((audio.get('artist') or [None])[0] or '').lower().strip()
|
||||||
|
file_albumartist = (
|
||||||
|
(audio.get('albumartist') or audio.get('album_artist') or [None])[0]
|
||||||
|
or ''
|
||||||
|
).lower().strip()
|
||||||
if file_title:
|
if file_title:
|
||||||
|
file_artists = [a for a in (file_artist, file_albumartist) if a]
|
||||||
|
if not file_artists:
|
||||||
|
file_artists = ['']
|
||||||
# Exact match first (fast path)
|
# Exact match first (fast path)
|
||||||
if (file_title, file_artist) in known_titles:
|
if any((file_title, artist) in known_titles for artist in file_artists):
|
||||||
is_known = True
|
is_known = True
|
||||||
else:
|
else:
|
||||||
# Normalized match: strip (feat. X), [FLAC 16bit], etc.
|
# Normalized match: strip (feat. X), [FLAC 16bit], etc.
|
||||||
clean_title = _strip_extras(file_title)
|
clean_title = _strip_extras(file_title)
|
||||||
clean_artist = _strip_extras(file_artist)
|
clean_artist = _strip_extras(file_artists[0])
|
||||||
# Also try first artist only (handles "Gorillaz, Dennis Hopper" → "Gorillaz")
|
# Also try first artist only (handles "Gorillaz, Dennis Hopper" → "Gorillaz")
|
||||||
first_artist = clean_artist.split(',')[0].strip() if clean_artist else ''
|
first_artist = clean_artist.split(',')[0].strip() if clean_artist else ''
|
||||||
if clean_title and (
|
if clean_title and (
|
||||||
|
|
@ -159,6 +172,16 @@ class OrphanFileDetectorJob(RepairJob):
|
||||||
(first_artist and (clean_title, first_artist) in known_titles_clean)
|
(first_artist and (clean_title, first_artist) in known_titles_clean)
|
||||||
):
|
):
|
||||||
is_known = True
|
is_known = True
|
||||||
|
if clean_title and not is_known:
|
||||||
|
for artist in file_artists[1:]:
|
||||||
|
clean_artist = _strip_extras(artist)
|
||||||
|
first_artist = clean_artist.split(',')[0].strip() if clean_artist else ''
|
||||||
|
if (
|
||||||
|
(clean_title, clean_artist) in known_titles_clean or
|
||||||
|
(first_artist and (clean_title, first_artist) in known_titles_clean)
|
||||||
|
):
|
||||||
|
is_known = True
|
||||||
|
break
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.debug("tag-based orphan check: %s", e)
|
logger.debug("tag-based orphan check: %s", e)
|
||||||
|
|
||||||
|
|
|
||||||
81
tests/test_orphan_file_detector.py
Normal file
81
tests/test_orphan_file_detector.py
Normal file
|
|
@ -0,0 +1,81 @@
|
||||||
|
"""Regression tests for the orphan file detector."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import sqlite3
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from core.repair_jobs.base import JobContext
|
||||||
|
from core.repair_jobs.orphan_file_detector import OrphanFileDetectorJob
|
||||||
|
|
||||||
|
|
||||||
|
class _DB:
|
||||||
|
def __init__(self, path: Path) -> None:
|
||||||
|
self.path = path
|
||||||
|
|
||||||
|
def _get_connection(self):
|
||||||
|
return sqlite3.connect(self.path)
|
||||||
|
|
||||||
|
|
||||||
|
def _seed_library(db_path: Path) -> None:
|
||||||
|
conn = sqlite3.connect(db_path)
|
||||||
|
try:
|
||||||
|
conn.executescript(
|
||||||
|
"""
|
||||||
|
CREATE TABLE artists (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
name TEXT NOT NULL
|
||||||
|
);
|
||||||
|
CREATE TABLE albums (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
artist_id INTEGER NOT NULL,
|
||||||
|
title TEXT NOT NULL
|
||||||
|
);
|
||||||
|
CREATE TABLE tracks (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
album_id INTEGER NOT NULL,
|
||||||
|
artist_id INTEGER NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
file_path TEXT
|
||||||
|
);
|
||||||
|
INSERT INTO artists (id, name) VALUES
|
||||||
|
(1, 'Clouddead89'),
|
||||||
|
(2, 'Featured Artist');
|
||||||
|
INSERT INTO albums (id, artist_id, title) VALUES
|
||||||
|
(10, 1, 'Perfect Match Error');
|
||||||
|
INSERT INTO tracks (id, album_id, artist_id, title, file_path) VALUES
|
||||||
|
(100, 10, 2, 'Perfect Match', '/old/prefix/elsewhere.mp3');
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
|
||||||
|
def test_orphan_detector_accepts_picard_albumartist_folder_match(tmp_path: Path) -> None:
|
||||||
|
"""Picard paths use albumartist/album (year)/track - title.
|
||||||
|
|
||||||
|
Even when the DB track artist is a featured artist, the album artist
|
||||||
|
folder should be enough to recognize the file as tracked.
|
||||||
|
"""
|
||||||
|
db_path = tmp_path / "library.sqlite"
|
||||||
|
_seed_library(db_path)
|
||||||
|
|
||||||
|
transfer = tmp_path / "Clouddead89" / "Perfect Match Error (2026)"
|
||||||
|
transfer.mkdir(parents=True)
|
||||||
|
audio_path = transfer / "01 - Perfect Match.mp3"
|
||||||
|
audio_path.write_bytes(b"not a real mp3; filename fallback handles this")
|
||||||
|
|
||||||
|
findings = []
|
||||||
|
context = JobContext(
|
||||||
|
db=_DB(db_path),
|
||||||
|
transfer_folder=str(tmp_path),
|
||||||
|
config_manager=None,
|
||||||
|
create_finding=lambda **kwargs: findings.append(kwargs) or True,
|
||||||
|
)
|
||||||
|
|
||||||
|
result = OrphanFileDetectorJob().scan(context)
|
||||||
|
|
||||||
|
assert result.scanned == 1
|
||||||
|
assert result.findings_created == 0
|
||||||
|
assert findings == []
|
||||||
Loading…
Reference in a new issue