Compare commits

...

4 Commits

Author SHA1 Message Date
Philip Guzman 5a2edb8b94 Add explainable matching engine 2026-06-30 17:42:33 -07:00
Philip Guzman 29e53f7dc6 Merge feature/logging into develop 2026-06-30 17:40:57 -07:00
Philip Guzman 4fde93613f Add opt-in diagnostic logging 2026-06-30 17:36:58 -07:00
Philip Guzman e4d5a32de2 Merge feature/configuration into develop 2026-06-30 17:35:59 -07:00
12 changed files with 359 additions and 4 deletions
+2
View File
@@ -12,6 +12,8 @@
- [x] Sample library fixtures
- [ ] Database V2 read-only parser
- [x] Configuration
- [x] Logging
- [x] Matching engine
## v0.2 — Diagnostics
+25
View File
@@ -0,0 +1,25 @@
# Application Logging
## Problem
The CLI reports final counts but provides no diagnostic trail when a scan behaves
unexpectedly. Troubleshooting should not require adding print statements or expose
library contents by default.
## Architecture
The project uses an isolated standard-library logger. It has no visible output by
default. `--verbose` writes progress to standard error, while `--log-file PATH`
writes an informational audit trail. Normal result lines remain on standard output.
## Edge Cases
- Reconfiguring logging in the same process must not duplicate handlers.
- Console and file logging may be enabled together.
- Log messages contain aggregate counts, not track names or crate contents.
- A default scan must remain quiet except for its established result output.
## Verification
Tests verify quiet defaults, file output, handler replacement, and unchanged CLI
result lines. The full sample-library scan remains read-only.
+29
View File
@@ -0,0 +1,29 @@
# Explainable Matching Engine
## Problem
Missing references need ranked candidate files, but a filename-only yes/no check
cannot explain ambiguity or cloud-provider conflict names.
## Architecture
The read-only matching engine indexes normalized filenames and scores only related
candidates. Every score contains evidence for filename, extension, and parent
folder. Exact filenames earn 60 points, normalized names 55, numeric conflict-name
matches 50, extensions 10, and parent folders 20.
The displayed percentage is an evidence score, not a statistical probability.
Metadata, duration, hashes, and fingerprints can add stronger evidence later.
## Edge Cases
- Unicode and case differences.
- OneDrive-style names such as `Track 2.mp3`.
- Duplicate candidates in different folders.
- Legitimate numbered song titles, which remain candidates but are never repaired.
- Unrelated names, which are not emitted as candidates.
## Verification
Tests cover exact, normalized, conflict-suffix, ambiguous, and unrelated filenames.
Candidate ordering is deterministic. The engine never changes a track or reference.
+15
View File
@@ -3,6 +3,7 @@ import argparse
from serato_doctor.config import ScanConfig
from serato_doctor.crate_parser import parse_crates
from serato_doctor.logging import configure_logging
from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_audio
from serato_doctor.report import write_csv, write_missing_report
@@ -20,6 +21,10 @@ def main():
default=[],
help="Old library root stored in crates; may be supplied more than once",
)
parser.add_argument(
"--verbose", action="store_true", help="Write diagnostic progress to stderr"
)
parser.add_argument("--log-file", help="Write scan progress to a log file")
args = parser.parse_args()
config = ScanConfig.build(
@@ -28,7 +33,13 @@ def main():
out=Path(args.out),
report=Path(args.report),
reference_roots=(Path(root) for root in args.reference_root),
verbose=args.verbose,
log_file=Path(args.log_file) if args.log_file else None,
)
logger = configure_logging(config.verbose, config.log_file)
logger.info("Starting read-only library scan")
logger.debug("Serato directory: %s", config.serato)
logger.debug("Music directory: %s", config.music)
library = Library.build(
references=parse_crates(
@@ -38,9 +49,13 @@ def main():
)
results = library.reconcile_by_filename()
missing_count = sum(1 for result in results if not result.exists_by_filename)
logger.info("Parsed %d crate references", len(library.references))
logger.info("Scanned %d disk tracks", len(library.tracks))
logger.info("Found %d references missing by filename", missing_count)
write_csv(results, config.out)
write_missing_report(results, config.report)
logger.info("Wrote CSV and missing-reference reports")
print(f"Crate references: {len(library.references)}")
print(f"Disk tracks: {len(library.tracks)}")
+14 -2
View File
@@ -1,6 +1,6 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Tuple
from typing import Iterable, Optional, Tuple
@dataclass(frozen=True)
@@ -12,6 +12,8 @@ class ScanConfig:
out: Path
report: Path
reference_roots: Tuple[Path, ...] = ()
verbose: bool = False
log_file: Optional[Path] = None
@classmethod
def build(
@@ -21,5 +23,15 @@ class ScanConfig:
out: Path,
report: Path,
reference_roots: Iterable[Path] = (),
verbose: bool = False,
log_file: Optional[Path] = None,
) -> "ScanConfig":
return cls(serato, music, out, report, tuple(reference_roots))
return cls(
serato,
music,
out,
report,
tuple(reference_roots),
verbose,
log_file,
)
+40
View File
@@ -0,0 +1,40 @@
import logging
from pathlib import Path
from typing import Optional
LOGGER_NAME = "serato_doctor"
LOG_FORMAT = "%(asctime)s %(levelname)s %(message)s"
def configure_logging(
verbose: bool = False, log_file: Optional[Path] = None
) -> logging.Logger:
"""Configure isolated application logging and return the project logger."""
logger = logging.getLogger(LOGGER_NAME)
logger.setLevel(logging.DEBUG)
logger.propagate = False
for handler in logger.handlers[:]:
handler.close()
logger.removeHandler(handler)
formatter = logging.Formatter(LOG_FORMAT)
if verbose:
console = logging.StreamHandler()
console.setLevel(logging.DEBUG)
console.setFormatter(formatter)
logger.addHandler(console)
if log_file is not None:
file_handler = logging.FileHandler(log_file, encoding="utf-8")
file_handler.setLevel(logging.INFO)
file_handler.setFormatter(formatter)
logger.addHandler(file_handler)
if not logger.handlers:
logger.addHandler(logging.NullHandler())
return logger
+86
View File
@@ -0,0 +1,86 @@
import re
import unicodedata
from collections import defaultdict
from pathlib import Path
from typing import DefaultDict, Iterable, List, Tuple
from serato_doctor.models.match import MatchEvidence, TrackMatch
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def normalize(value: str) -> str:
return unicodedata.normalize("NFKC", value).casefold()
def cloud_conflict_name(filename: str) -> str:
"""Remove a trailing numeric cloud-conflict suffix from a filename stem."""
path = Path(filename)
stem = re.sub(r" \d+$", "", path.stem)
return normalize(stem + path.suffix)
def score_candidate(reference: TrackReference, track: DiskTrack) -> TrackMatch:
reference_name = reference.filename
track_name = track.filename
if reference_name == track_name:
filename_points = 60
filename_reason = "Filename is identical"
elif normalize(reference_name) == normalize(track_name):
filename_points = 55
filename_reason = "Filename matches after case and Unicode normalization"
elif cloud_conflict_name(reference_name) == cloud_conflict_name(track_name):
filename_points = 50
filename_reason = "Filename matches after removing a numeric conflict suffix"
else:
filename_points = 0
filename_reason = "Filename does not match"
same_extension = normalize(reference.path.suffix) == normalize(track.suffix)
same_parent = normalize(reference.path.parent.name) == normalize(
track.path.parent.name
)
evidence = (
MatchEvidence(
"filename",
filename_points > 0,
filename_points,
60,
filename_reason,
),
MatchEvidence(
"extension",
same_extension,
10 if same_extension else 0,
10,
"File extension matches" if same_extension else "File extension differs",
),
MatchEvidence(
"parent_folder",
same_parent,
20 if same_parent else 0,
20,
"Parent folder matches" if same_parent else "Parent folder differs",
),
)
return TrackMatch(reference, track, evidence)
class MatchingEngine:
"""Find and rank filename-related disk candidates without modifying files."""
def __init__(self, tracks: Iterable[DiskTrack]):
self._by_conflict_name: DefaultDict[str, List[DiskTrack]] = defaultdict(list)
for track in tracks:
self._by_conflict_name[cloud_conflict_name(track.filename)].append(track)
def candidates_for(self, reference: TrackReference) -> Tuple[TrackMatch, ...]:
candidates = self._by_conflict_name.get(
cloud_conflict_name(reference.filename), []
)
matches = [score_candidate(reference, track) for track in candidates]
return tuple(
sorted(matches, key=lambda match: (-match.score, str(match.track.path)))
)
+10 -1
View File
@@ -1,6 +1,15 @@
from serato_doctor.models.crate import Crate
from serato_doctor.models.library import Library
from serato_doctor.models.match import MatchEvidence, TrackMatch
from serato_doctor.models.reference import ReferenceResult, TrackReference
from serato_doctor.models.track import DiskTrack
__all__ = ["Crate", "DiskTrack", "Library", "ReferenceResult", "TrackReference"]
__all__ = [
"Crate",
"DiskTrack",
"Library",
"MatchEvidence",
"ReferenceResult",
"TrackMatch",
"TrackReference",
]
+39
View File
@@ -0,0 +1,39 @@
from dataclasses import dataclass
from typing import Tuple
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
@dataclass(frozen=True)
class MatchEvidence:
"""One explainable scoring decision for a candidate track."""
field: str
matched: bool
points: int
max_points: int
explanation: str
@dataclass(frozen=True)
class TrackMatch:
"""A ranked candidate backed by explicit, inspectable evidence."""
reference: TrackReference
track: DiskTrack
evidence: Tuple[MatchEvidence, ...]
@property
def score(self) -> int:
return sum(item.points for item in self.evidence)
@property
def max_score(self) -> int:
return sum(item.max_points for item in self.evidence)
@property
def score_percent(self) -> float:
if not self.max_score:
return 0.0
return round(self.score / self.max_score * 100, 1)
+3 -1
View File
@@ -37,7 +37,9 @@ def test_cli_writes_reports_and_prints_counts(tmp_path, monkeypatch, capsys):
main()
output = capsys.readouterr().out
captured = capsys.readouterr()
output = captured.out
assert captured.err == ""
assert "Crate references: 2" in output
assert "Disk tracks: 1" in output
assert "Missing by filename: 1" in output
+34
View File
@@ -0,0 +1,34 @@
import logging
from serato_doctor.logging import LOGGER_NAME, configure_logging
def test_logging_is_quiet_by_default(capsys):
logger = configure_logging()
logger.info("not visible")
assert capsys.readouterr().err == ""
def test_logging_writes_aggregate_progress_to_file(tmp_path):
log_path = tmp_path / "scan.log"
logger = configure_logging(log_file=log_path)
logger.info("Scanned %d disk tracks", 10)
contents = log_path.read_text(encoding="utf-8")
assert "INFO Scanned 10 disk tracks" in contents
def test_reconfiguring_logging_replaces_handlers():
configure_logging(verbose=True)
logger = configure_logging(verbose=True)
active_handlers = [
handler
for handler in logger.handlers
if not isinstance(handler, logging.NullHandler)
]
assert logger.name == LOGGER_NAME
assert len(active_handlers) == 1
+62
View File
@@ -0,0 +1,62 @@
from pathlib import Path
from serato_doctor.matching import MatchingEngine, score_candidate
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def reference(filename, folder="House"):
return TrackReference(
Path("Test.crate"), Path("/old") / folder / filename, filename
)
def track(filename, folder="House"):
path = Path("/new") / folder / filename
return DiskTrack(path, filename, 100, path.suffix.lower())
def test_exact_candidate_has_full_evidence_score():
match = score_candidate(reference("Track.mp3"), track("Track.mp3"))
assert match.score == 90
assert match.max_score == 90
assert match.score_percent == 100.0
assert all(item.matched for item in match.evidence)
def test_cloud_conflict_suffix_is_explained():
match = score_candidate(reference("Track.mp3"), track("Track 2.mp3"))
assert match.score == 80
assert match.score_percent == 88.9
assert match.evidence[0].explanation == (
"Filename matches after removing a numeric conflict suffix"
)
def test_case_normalized_match_scores_below_exact():
match = score_candidate(reference("TRACK.MP3"), track("track.mp3"))
assert match.score == 85
assert match.evidence[0].points == 55
def test_engine_omits_unrelated_filenames():
engine = MatchingEngine([track("Different.mp3")])
assert engine.candidates_for(reference("Missing.mp3")) == ()
def test_ambiguous_candidates_are_ranked_deterministically():
engine = MatchingEngine(
[track("Track 2.mp3", "Other"), track("Track 3.mp3", "House")]
)
matches = engine.candidates_for(reference("Track.mp3"))
assert [match.track.filename for match in matches] == [
"Track 3.mp3",
"Track 2.mp3",
]
assert [match.score for match in matches] == [80, 60]