diff --git a/ROADMAP.md b/ROADMAP.md index f386d71..1ec0b14 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -23,7 +23,7 @@ - [ ] Orphaned audio detection - [ ] OneDrive rename detection - [ ] Crate classification: static vs smart/dynamic -- [ ] Library health score +- [x] Library health score ## v0.3 — Safe Repair diff --git a/docs/design/health-engine.md b/docs/design/health-engine.md new file mode 100644 index 0000000..040d25a --- /dev/null +++ b/docs/design/health-engine.md @@ -0,0 +1,31 @@ +# Library Health Engine + +## Problem + +Raw missing-reference counts do not provide a compact view of library integrity, +but an opaque blended score would imply confidence the current data cannot support. + +## Architecture + +The health engine produces an immutable report from the core `Library`. Its score +is only the percentage of crate references resolved by exact filename. The report +also exposes missing references, unique missing filenames, duplicate filename +groups, extra duplicate files, unused tracks, and missing references with matching +candidates. + +Duplicate, unused, and candidate counts are informational. They do not affect the +score until the project has a documented and validated weighting policy. An empty +library has no score rather than a misleading 0% or 100%. + +## Edge Cases + +- The same missing filename referenced by several crates. +- Several disk files sharing a filename. +- Conflict-suffixed files that are unused but may be match candidates. +- Libraries with no crate references. +- Tracks referenced by filename from more than one crate. + +## Verification + +Tests assert every metric, the disclosed score basis, candidate integration, and +empty-library behavior. Analysis remains entirely read-only. diff --git a/serato_doctor/health.py b/serato_doctor/health.py new file mode 100644 index 0000000..e708707 --- /dev/null +++ b/serato_doctor/health.py @@ -0,0 +1,43 @@ +from collections import Counter + +from serato_doctor.matching import MatchingEngine +from serato_doctor.models.health import HealthReport +from serato_doctor.models.library import Library + + +def analyze_health(library: Library) -> HealthReport: + """Calculate defensible health metrics without changing the library.""" + + results = library.reconcile_by_filename() + missing = [result for result in results if not result.exists_by_filename] + healthy_count = len(results) - len(missing) + score = ( + round(healthy_count / len(results) * 100, 1) if results else None + ) + + disk_name_counts = Counter(track.filename for track in library.tracks) + duplicate_counts = [count for count in disk_name_counts.values() if count > 1] + referenced_names = {reference.filename for reference in library.references} + unused_count = sum( + 1 for track in library.tracks if track.filename not in referenced_names + ) + + matcher = MatchingEngine(library.tracks) + suggested_count = sum( + bool(matcher.candidates_for(result.reference)) for result in missing + ) + + return HealthReport( + score=score, + total_references=len(results), + healthy_references=healthy_count, + missing_references=len(missing), + unique_missing_filenames=len( + {result.reference.filename for result in missing} + ), + disk_tracks=len(library.tracks), + duplicate_filename_groups=len(duplicate_counts), + duplicate_files=sum(count - 1 for count in duplicate_counts), + unused_tracks=unused_count, + suggested_matches=suggested_count, + ) diff --git a/serato_doctor/models/__init__.py b/serato_doctor/models/__init__.py index 217820c..39d2ebc 100644 --- a/serato_doctor/models/__init__.py +++ b/serato_doctor/models/__init__.py @@ -1,4 +1,5 @@ from serato_doctor.models.crate import Crate +from serato_doctor.models.health import HealthReport from serato_doctor.models.library import Library from serato_doctor.models.match import MatchEvidence, TrackMatch from serato_doctor.models.reference import ReferenceResult, TrackReference @@ -7,6 +8,7 @@ from serato_doctor.models.track import DiskTrack __all__ = [ "Crate", "DiskTrack", + "HealthReport", "Library", "MatchEvidence", "ReferenceResult", diff --git a/serato_doctor/models/health.py b/serato_doctor/models/health.py new file mode 100644 index 0000000..84c0b77 --- /dev/null +++ b/serato_doctor/models/health.py @@ -0,0 +1,22 @@ +from dataclasses import dataclass +from typing import Optional + + +@dataclass(frozen=True) +class HealthReport: + """Transparent aggregate findings from a read-only library analysis.""" + + score: Optional[float] + total_references: int + healthy_references: int + missing_references: int + unique_missing_filenames: int + disk_tracks: int + duplicate_filename_groups: int + duplicate_files: int + unused_tracks: int + suggested_matches: int + + @property + def score_basis(self) -> str: + return "Resolved crate references / total crate references" diff --git a/tests/test_health.py b/tests/test_health.py new file mode 100644 index 0000000..5e4dc8c --- /dev/null +++ b/tests/test_health.py @@ -0,0 +1,56 @@ +from pathlib import Path + +from serato_doctor.health import analyze_health +from serato_doctor.models.library import Library +from serato_doctor.models.reference import TrackReference +from serato_doctor.models.track import DiskTrack + + +def reference(filename): + return TrackReference( + Path("Test.crate"), Path("/old/House") / filename, filename + ) + + +def track(filename, folder="House"): + path = Path("/new") / folder / filename + return DiskTrack(path, filename, 100, path.suffix.lower()) + + +def test_health_report_exposes_each_metric(): + library = Library.build( + [ + reference("Found.mp3"), + reference("Conflict.mp3"), + reference("Absent.mp3"), + ], + [ + track("Found.mp3"), + track("Conflict 2.mp3"), + track("Conflict 2.mp3", "Backup"), + track("Unused.mp3"), + ], + ) + + report = analyze_health(library) + + assert report.score == 33.3 + assert report.score_basis == ( + "Resolved crate references / total crate references" + ) + assert report.total_references == 3 + assert report.healthy_references == 1 + assert report.missing_references == 2 + assert report.unique_missing_filenames == 2 + assert report.disk_tracks == 4 + assert report.duplicate_filename_groups == 1 + assert report.duplicate_files == 1 + assert report.unused_tracks == 3 + assert report.suggested_matches == 1 + + +def test_empty_library_has_no_health_score(): + report = analyze_health(Library.build([], [])) + + assert report.score is None + assert report.total_references == 0