Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 22424d0b5c | |||
| b8bf9ec6e1 | |||
| 167f029c28 | |||
| 5a2edb8b94 | |||
| 29e53f7dc6 |
+2
-1
@@ -13,6 +13,7 @@
|
|||||||
- [ ] Database V2 read-only parser
|
- [ ] Database V2 read-only parser
|
||||||
- [x] Configuration
|
- [x] Configuration
|
||||||
- [x] Logging
|
- [x] Logging
|
||||||
|
- [x] Matching engine
|
||||||
|
|
||||||
## v0.2 — Diagnostics
|
## v0.2 — Diagnostics
|
||||||
|
|
||||||
@@ -22,7 +23,7 @@
|
|||||||
- [ ] Orphaned audio detection
|
- [ ] Orphaned audio detection
|
||||||
- [ ] OneDrive rename detection
|
- [ ] OneDrive rename detection
|
||||||
- [ ] Crate classification: static vs smart/dynamic
|
- [ ] Crate classification: static vs smart/dynamic
|
||||||
- [ ] Library health score
|
- [x] Library health score
|
||||||
|
|
||||||
## v0.3 — Safe Repair
|
## v0.3 — Safe Repair
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
# Library Health Engine
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
Raw missing-reference counts do not provide a compact view of library integrity,
|
||||||
|
but an opaque blended score would imply confidence the current data cannot support.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
The health engine produces an immutable report from the core `Library`. Its score
|
||||||
|
is only the percentage of crate references resolved by exact filename. The report
|
||||||
|
also exposes missing references, unique missing filenames, duplicate filename
|
||||||
|
groups, extra duplicate files, unused tracks, and missing references with matching
|
||||||
|
candidates.
|
||||||
|
|
||||||
|
Duplicate, unused, and candidate counts are informational. They do not affect the
|
||||||
|
score until the project has a documented and validated weighting policy. An empty
|
||||||
|
library has no score rather than a misleading 0% or 100%.
|
||||||
|
|
||||||
|
## Edge Cases
|
||||||
|
|
||||||
|
- The same missing filename referenced by several crates.
|
||||||
|
- Several disk files sharing a filename.
|
||||||
|
- Conflict-suffixed files that are unused but may be match candidates.
|
||||||
|
- Libraries with no crate references.
|
||||||
|
- Tracks referenced by filename from more than one crate.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
Tests assert every metric, the disclosed score basis, candidate integration, and
|
||||||
|
empty-library behavior. Analysis remains entirely read-only.
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
# Explainable Matching Engine
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
Missing references need ranked candidate files, but a filename-only yes/no check
|
||||||
|
cannot explain ambiguity or cloud-provider conflict names.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
The read-only matching engine indexes normalized filenames and scores only related
|
||||||
|
candidates. Every score contains evidence for filename, extension, and parent
|
||||||
|
folder. Exact filenames earn 60 points, normalized names 55, numeric conflict-name
|
||||||
|
matches 50, extensions 10, and parent folders 20.
|
||||||
|
|
||||||
|
The displayed percentage is an evidence score, not a statistical probability.
|
||||||
|
Metadata, duration, hashes, and fingerprints can add stronger evidence later.
|
||||||
|
|
||||||
|
## Edge Cases
|
||||||
|
|
||||||
|
- Unicode and case differences.
|
||||||
|
- OneDrive-style names such as `Track 2.mp3`.
|
||||||
|
- Duplicate candidates in different folders.
|
||||||
|
- Legitimate numbered song titles, which remain candidates but are never repaired.
|
||||||
|
- Unrelated names, which are not emitted as candidates.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
Tests cover exact, normalized, conflict-suffix, ambiguous, and unrelated filenames.
|
||||||
|
Candidate ordering is deterministic. The engine never changes a track or reference.
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
from serato_doctor.matching import MatchingEngine
|
||||||
|
from serato_doctor.models.health import HealthReport
|
||||||
|
from serato_doctor.models.library import Library
|
||||||
|
|
||||||
|
|
||||||
|
def analyze_health(library: Library) -> HealthReport:
|
||||||
|
"""Calculate defensible health metrics without changing the library."""
|
||||||
|
|
||||||
|
results = library.reconcile_by_filename()
|
||||||
|
missing = [result for result in results if not result.exists_by_filename]
|
||||||
|
healthy_count = len(results) - len(missing)
|
||||||
|
score = (
|
||||||
|
round(healthy_count / len(results) * 100, 1) if results else None
|
||||||
|
)
|
||||||
|
|
||||||
|
disk_name_counts = Counter(track.filename for track in library.tracks)
|
||||||
|
duplicate_counts = [count for count in disk_name_counts.values() if count > 1]
|
||||||
|
referenced_names = {reference.filename for reference in library.references}
|
||||||
|
unused_count = sum(
|
||||||
|
1 for track in library.tracks if track.filename not in referenced_names
|
||||||
|
)
|
||||||
|
|
||||||
|
matcher = MatchingEngine(library.tracks)
|
||||||
|
suggested_count = sum(
|
||||||
|
bool(matcher.candidates_for(result.reference)) for result in missing
|
||||||
|
)
|
||||||
|
|
||||||
|
return HealthReport(
|
||||||
|
score=score,
|
||||||
|
total_references=len(results),
|
||||||
|
healthy_references=healthy_count,
|
||||||
|
missing_references=len(missing),
|
||||||
|
unique_missing_filenames=len(
|
||||||
|
{result.reference.filename for result in missing}
|
||||||
|
),
|
||||||
|
disk_tracks=len(library.tracks),
|
||||||
|
duplicate_filename_groups=len(duplicate_counts),
|
||||||
|
duplicate_files=sum(count - 1 for count in duplicate_counts),
|
||||||
|
unused_tracks=unused_count,
|
||||||
|
suggested_matches=suggested_count,
|
||||||
|
)
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
import re
|
||||||
|
import unicodedata
|
||||||
|
from collections import defaultdict
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import DefaultDict, Iterable, List, Tuple
|
||||||
|
|
||||||
|
from serato_doctor.models.match import MatchEvidence, TrackMatch
|
||||||
|
from serato_doctor.models.reference import TrackReference
|
||||||
|
from serato_doctor.models.track import DiskTrack
|
||||||
|
|
||||||
|
|
||||||
|
def normalize(value: str) -> str:
|
||||||
|
return unicodedata.normalize("NFKC", value).casefold()
|
||||||
|
|
||||||
|
|
||||||
|
def cloud_conflict_name(filename: str) -> str:
|
||||||
|
"""Remove a trailing numeric cloud-conflict suffix from a filename stem."""
|
||||||
|
|
||||||
|
path = Path(filename)
|
||||||
|
stem = re.sub(r" \d+$", "", path.stem)
|
||||||
|
return normalize(stem + path.suffix)
|
||||||
|
|
||||||
|
|
||||||
|
def score_candidate(reference: TrackReference, track: DiskTrack) -> TrackMatch:
|
||||||
|
reference_name = reference.filename
|
||||||
|
track_name = track.filename
|
||||||
|
|
||||||
|
if reference_name == track_name:
|
||||||
|
filename_points = 60
|
||||||
|
filename_reason = "Filename is identical"
|
||||||
|
elif normalize(reference_name) == normalize(track_name):
|
||||||
|
filename_points = 55
|
||||||
|
filename_reason = "Filename matches after case and Unicode normalization"
|
||||||
|
elif cloud_conflict_name(reference_name) == cloud_conflict_name(track_name):
|
||||||
|
filename_points = 50
|
||||||
|
filename_reason = "Filename matches after removing a numeric conflict suffix"
|
||||||
|
else:
|
||||||
|
filename_points = 0
|
||||||
|
filename_reason = "Filename does not match"
|
||||||
|
|
||||||
|
same_extension = normalize(reference.path.suffix) == normalize(track.suffix)
|
||||||
|
same_parent = normalize(reference.path.parent.name) == normalize(
|
||||||
|
track.path.parent.name
|
||||||
|
)
|
||||||
|
evidence = (
|
||||||
|
MatchEvidence(
|
||||||
|
"filename",
|
||||||
|
filename_points > 0,
|
||||||
|
filename_points,
|
||||||
|
60,
|
||||||
|
filename_reason,
|
||||||
|
),
|
||||||
|
MatchEvidence(
|
||||||
|
"extension",
|
||||||
|
same_extension,
|
||||||
|
10 if same_extension else 0,
|
||||||
|
10,
|
||||||
|
"File extension matches" if same_extension else "File extension differs",
|
||||||
|
),
|
||||||
|
MatchEvidence(
|
||||||
|
"parent_folder",
|
||||||
|
same_parent,
|
||||||
|
20 if same_parent else 0,
|
||||||
|
20,
|
||||||
|
"Parent folder matches" if same_parent else "Parent folder differs",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
return TrackMatch(reference, track, evidence)
|
||||||
|
|
||||||
|
|
||||||
|
class MatchingEngine:
|
||||||
|
"""Find and rank filename-related disk candidates without modifying files."""
|
||||||
|
|
||||||
|
def __init__(self, tracks: Iterable[DiskTrack]):
|
||||||
|
self._by_conflict_name: DefaultDict[str, List[DiskTrack]] = defaultdict(list)
|
||||||
|
for track in tracks:
|
||||||
|
self._by_conflict_name[cloud_conflict_name(track.filename)].append(track)
|
||||||
|
|
||||||
|
def candidates_for(self, reference: TrackReference) -> Tuple[TrackMatch, ...]:
|
||||||
|
candidates = self._by_conflict_name.get(
|
||||||
|
cloud_conflict_name(reference.filename), []
|
||||||
|
)
|
||||||
|
matches = [score_candidate(reference, track) for track in candidates]
|
||||||
|
return tuple(
|
||||||
|
sorted(matches, key=lambda match: (-match.score, str(match.track.path)))
|
||||||
|
)
|
||||||
@@ -1,6 +1,17 @@
|
|||||||
from serato_doctor.models.crate import Crate
|
from serato_doctor.models.crate import Crate
|
||||||
|
from serato_doctor.models.health import HealthReport
|
||||||
from serato_doctor.models.library import Library
|
from serato_doctor.models.library import Library
|
||||||
|
from serato_doctor.models.match import MatchEvidence, TrackMatch
|
||||||
from serato_doctor.models.reference import ReferenceResult, TrackReference
|
from serato_doctor.models.reference import ReferenceResult, TrackReference
|
||||||
from serato_doctor.models.track import DiskTrack
|
from serato_doctor.models.track import DiskTrack
|
||||||
|
|
||||||
__all__ = ["Crate", "DiskTrack", "Library", "ReferenceResult", "TrackReference"]
|
__all__ = [
|
||||||
|
"Crate",
|
||||||
|
"DiskTrack",
|
||||||
|
"HealthReport",
|
||||||
|
"Library",
|
||||||
|
"MatchEvidence",
|
||||||
|
"ReferenceResult",
|
||||||
|
"TrackMatch",
|
||||||
|
"TrackReference",
|
||||||
|
]
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class HealthReport:
|
||||||
|
"""Transparent aggregate findings from a read-only library analysis."""
|
||||||
|
|
||||||
|
score: Optional[float]
|
||||||
|
total_references: int
|
||||||
|
healthy_references: int
|
||||||
|
missing_references: int
|
||||||
|
unique_missing_filenames: int
|
||||||
|
disk_tracks: int
|
||||||
|
duplicate_filename_groups: int
|
||||||
|
duplicate_files: int
|
||||||
|
unused_tracks: int
|
||||||
|
suggested_matches: int
|
||||||
|
|
||||||
|
@property
|
||||||
|
def score_basis(self) -> str:
|
||||||
|
return "Resolved crate references / total crate references"
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Tuple
|
||||||
|
|
||||||
|
from serato_doctor.models.reference import TrackReference
|
||||||
|
from serato_doctor.models.track import DiskTrack
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class MatchEvidence:
|
||||||
|
"""One explainable scoring decision for a candidate track."""
|
||||||
|
|
||||||
|
field: str
|
||||||
|
matched: bool
|
||||||
|
points: int
|
||||||
|
max_points: int
|
||||||
|
explanation: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class TrackMatch:
|
||||||
|
"""A ranked candidate backed by explicit, inspectable evidence."""
|
||||||
|
|
||||||
|
reference: TrackReference
|
||||||
|
track: DiskTrack
|
||||||
|
evidence: Tuple[MatchEvidence, ...]
|
||||||
|
|
||||||
|
@property
|
||||||
|
def score(self) -> int:
|
||||||
|
return sum(item.points for item in self.evidence)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def max_score(self) -> int:
|
||||||
|
return sum(item.max_points for item in self.evidence)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def score_percent(self) -> float:
|
||||||
|
if not self.max_score:
|
||||||
|
return 0.0
|
||||||
|
return round(self.score / self.max_score * 100, 1)
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from serato_doctor.health import analyze_health
|
||||||
|
from serato_doctor.models.library import Library
|
||||||
|
from serato_doctor.models.reference import TrackReference
|
||||||
|
from serato_doctor.models.track import DiskTrack
|
||||||
|
|
||||||
|
|
||||||
|
def reference(filename):
|
||||||
|
return TrackReference(
|
||||||
|
Path("Test.crate"), Path("/old/House") / filename, filename
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def track(filename, folder="House"):
|
||||||
|
path = Path("/new") / folder / filename
|
||||||
|
return DiskTrack(path, filename, 100, path.suffix.lower())
|
||||||
|
|
||||||
|
|
||||||
|
def test_health_report_exposes_each_metric():
|
||||||
|
library = Library.build(
|
||||||
|
[
|
||||||
|
reference("Found.mp3"),
|
||||||
|
reference("Conflict.mp3"),
|
||||||
|
reference("Absent.mp3"),
|
||||||
|
],
|
||||||
|
[
|
||||||
|
track("Found.mp3"),
|
||||||
|
track("Conflict 2.mp3"),
|
||||||
|
track("Conflict 2.mp3", "Backup"),
|
||||||
|
track("Unused.mp3"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
report = analyze_health(library)
|
||||||
|
|
||||||
|
assert report.score == 33.3
|
||||||
|
assert report.score_basis == (
|
||||||
|
"Resolved crate references / total crate references"
|
||||||
|
)
|
||||||
|
assert report.total_references == 3
|
||||||
|
assert report.healthy_references == 1
|
||||||
|
assert report.missing_references == 2
|
||||||
|
assert report.unique_missing_filenames == 2
|
||||||
|
assert report.disk_tracks == 4
|
||||||
|
assert report.duplicate_filename_groups == 1
|
||||||
|
assert report.duplicate_files == 1
|
||||||
|
assert report.unused_tracks == 3
|
||||||
|
assert report.suggested_matches == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_library_has_no_health_score():
|
||||||
|
report = analyze_health(Library.build([], []))
|
||||||
|
|
||||||
|
assert report.score is None
|
||||||
|
assert report.total_references == 0
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from serato_doctor.matching import MatchingEngine, score_candidate
|
||||||
|
from serato_doctor.models.reference import TrackReference
|
||||||
|
from serato_doctor.models.track import DiskTrack
|
||||||
|
|
||||||
|
|
||||||
|
def reference(filename, folder="House"):
|
||||||
|
return TrackReference(
|
||||||
|
Path("Test.crate"), Path("/old") / folder / filename, filename
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def track(filename, folder="House"):
|
||||||
|
path = Path("/new") / folder / filename
|
||||||
|
return DiskTrack(path, filename, 100, path.suffix.lower())
|
||||||
|
|
||||||
|
|
||||||
|
def test_exact_candidate_has_full_evidence_score():
|
||||||
|
match = score_candidate(reference("Track.mp3"), track("Track.mp3"))
|
||||||
|
|
||||||
|
assert match.score == 90
|
||||||
|
assert match.max_score == 90
|
||||||
|
assert match.score_percent == 100.0
|
||||||
|
assert all(item.matched for item in match.evidence)
|
||||||
|
|
||||||
|
|
||||||
|
def test_cloud_conflict_suffix_is_explained():
|
||||||
|
match = score_candidate(reference("Track.mp3"), track("Track 2.mp3"))
|
||||||
|
|
||||||
|
assert match.score == 80
|
||||||
|
assert match.score_percent == 88.9
|
||||||
|
assert match.evidence[0].explanation == (
|
||||||
|
"Filename matches after removing a numeric conflict suffix"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_case_normalized_match_scores_below_exact():
|
||||||
|
match = score_candidate(reference("TRACK.MP3"), track("track.mp3"))
|
||||||
|
|
||||||
|
assert match.score == 85
|
||||||
|
assert match.evidence[0].points == 55
|
||||||
|
|
||||||
|
|
||||||
|
def test_engine_omits_unrelated_filenames():
|
||||||
|
engine = MatchingEngine([track("Different.mp3")])
|
||||||
|
|
||||||
|
assert engine.candidates_for(reference("Missing.mp3")) == ()
|
||||||
|
|
||||||
|
|
||||||
|
def test_ambiguous_candidates_are_ranked_deterministically():
|
||||||
|
engine = MatchingEngine(
|
||||||
|
[track("Track 2.mp3", "Other"), track("Track 3.mp3", "House")]
|
||||||
|
)
|
||||||
|
|
||||||
|
matches = engine.candidates_for(reference("Track.mp3"))
|
||||||
|
|
||||||
|
assert [match.track.filename for match in matches] == [
|
||||||
|
"Track 3.mp3",
|
||||||
|
"Track 2.mp3",
|
||||||
|
]
|
||||||
|
assert [match.score for match in matches] == [80, 60]
|
||||||
Reference in New Issue
Block a user