Compare commits

..

2 Commits

Author SHA1 Message Date
Philip Guzman 22424d0b5c Merge feature/health-engine into develop 2026-06-30 17:54:20 -07:00
Philip Guzman b8bf9ec6e1 Add transparent library health engine 2026-06-30 17:47:54 -07:00
6 changed files with 155 additions and 1 deletions
+1 -1
View File
@@ -23,7 +23,7 @@
- [ ] Orphaned audio detection
- [ ] OneDrive rename detection
- [ ] Crate classification: static vs smart/dynamic
- [ ] Library health score
- [x] Library health score
## v0.3 — Safe Repair
+31
View File
@@ -0,0 +1,31 @@
# Library Health Engine
## Problem
Raw missing-reference counts do not provide a compact view of library integrity,
but an opaque blended score would imply confidence the current data cannot support.
## Architecture
The health engine produces an immutable report from the core `Library`. Its score
is only the percentage of crate references resolved by exact filename. The report
also exposes missing references, unique missing filenames, duplicate filename
groups, extra duplicate files, unused tracks, and missing references with matching
candidates.
Duplicate, unused, and candidate counts are informational. They do not affect the
score until the project has a documented and validated weighting policy. An empty
library has no score rather than a misleading 0% or 100%.
## Edge Cases
- The same missing filename referenced by several crates.
- Several disk files sharing a filename.
- Conflict-suffixed files that are unused but may be match candidates.
- Libraries with no crate references.
- Tracks referenced by filename from more than one crate.
## Verification
Tests assert every metric, the disclosed score basis, candidate integration, and
empty-library behavior. Analysis remains entirely read-only.
+43
View File
@@ -0,0 +1,43 @@
from collections import Counter
from serato_doctor.matching import MatchingEngine
from serato_doctor.models.health import HealthReport
from serato_doctor.models.library import Library
def analyze_health(library: Library) -> HealthReport:
"""Calculate defensible health metrics without changing the library."""
results = library.reconcile_by_filename()
missing = [result for result in results if not result.exists_by_filename]
healthy_count = len(results) - len(missing)
score = (
round(healthy_count / len(results) * 100, 1) if results else None
)
disk_name_counts = Counter(track.filename for track in library.tracks)
duplicate_counts = [count for count in disk_name_counts.values() if count > 1]
referenced_names = {reference.filename for reference in library.references}
unused_count = sum(
1 for track in library.tracks if track.filename not in referenced_names
)
matcher = MatchingEngine(library.tracks)
suggested_count = sum(
bool(matcher.candidates_for(result.reference)) for result in missing
)
return HealthReport(
score=score,
total_references=len(results),
healthy_references=healthy_count,
missing_references=len(missing),
unique_missing_filenames=len(
{result.reference.filename for result in missing}
),
disk_tracks=len(library.tracks),
duplicate_filename_groups=len(duplicate_counts),
duplicate_files=sum(count - 1 for count in duplicate_counts),
unused_tracks=unused_count,
suggested_matches=suggested_count,
)
+2
View File
@@ -1,4 +1,5 @@
from serato_doctor.models.crate import Crate
from serato_doctor.models.health import HealthReport
from serato_doctor.models.library import Library
from serato_doctor.models.match import MatchEvidence, TrackMatch
from serato_doctor.models.reference import ReferenceResult, TrackReference
@@ -7,6 +8,7 @@ from serato_doctor.models.track import DiskTrack
__all__ = [
"Crate",
"DiskTrack",
"HealthReport",
"Library",
"MatchEvidence",
"ReferenceResult",
+22
View File
@@ -0,0 +1,22 @@
from dataclasses import dataclass
from typing import Optional
@dataclass(frozen=True)
class HealthReport:
"""Transparent aggregate findings from a read-only library analysis."""
score: Optional[float]
total_references: int
healthy_references: int
missing_references: int
unique_missing_filenames: int
disk_tracks: int
duplicate_filename_groups: int
duplicate_files: int
unused_tracks: int
suggested_matches: int
@property
def score_basis(self) -> str:
return "Resolved crate references / total crate references"
+56
View File
@@ -0,0 +1,56 @@
from pathlib import Path
from serato_doctor.health import analyze_health
from serato_doctor.models.library import Library
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def reference(filename):
return TrackReference(
Path("Test.crate"), Path("/old/House") / filename, filename
)
def track(filename, folder="House"):
path = Path("/new") / folder / filename
return DiskTrack(path, filename, 100, path.suffix.lower())
def test_health_report_exposes_each_metric():
library = Library.build(
[
reference("Found.mp3"),
reference("Conflict.mp3"),
reference("Absent.mp3"),
],
[
track("Found.mp3"),
track("Conflict 2.mp3"),
track("Conflict 2.mp3", "Backup"),
track("Unused.mp3"),
],
)
report = analyze_health(library)
assert report.score == 33.3
assert report.score_basis == (
"Resolved crate references / total crate references"
)
assert report.total_references == 3
assert report.healthy_references == 1
assert report.missing_references == 2
assert report.unique_missing_filenames == 2
assert report.disk_tracks == 4
assert report.duplicate_filename_groups == 1
assert report.duplicate_files == 1
assert report.unused_tracks == 3
assert report.suggested_matches == 1
def test_empty_library_has_no_health_score():
report = analyze_health(Library.build([], []))
assert report.score is None
assert report.total_references == 0