Merge feature/configuration into develop

This commit is contained in:
Philip Guzman
2026-06-30 17:35:59 -07:00
9 changed files with 211 additions and 38 deletions
+1
View File
@@ -11,6 +11,7 @@
- [x] Test suite - [x] Test suite
- [x] Sample library fixtures - [x] Sample library fixtures
- [ ] Database V2 read-only parser - [ ] Database V2 read-only parser
- [x] Configuration
## v0.2 — Diagnostics ## v0.2 — Diagnostics
+25
View File
@@ -0,0 +1,25 @@
# Scan Configuration
## Problem
The prototype recognized crate paths by matching one developer's historical
OneDrive path. That made otherwise valid libraries invisible to the parser.
## Architecture
`ScanConfig` owns the paths for a read-only scan. The CLI accepts repeatable
`--reference-root` options for roots embedded in crates before a migration. The
parser uses those roots as record boundaries. Without explicit roots, it recognizes
generic macOS home and mounted-volume paths without embedding a username.
## Edge Cases
- A library may have references from more than one historical root.
- Root paths may contain spaces or have leading/trailing separators.
- Existing `/Users/...` and `/Volumes/...` crates must work without new flags.
- A record without a recognized terminator remains ignored.
## Verification
Tests cover default root discovery, a custom migrated root, the CLI option, and
the existing synthetic sample library. The scanner remains read-only.
+1 -1
View File
@@ -7,7 +7,7 @@ from typing import Optional
SAMPLE_ROOT = Path(__file__).parent SAMPLE_ROOT = Path(__file__).parent
SERATO_PATH_PREFIX = "Users/djsplice/OneDrive/Jukebox/" SERATO_PATH_PREFIX = "Users/sample-user/OneDrive/Jukebox/"
def load_manifest() -> dict: def load_manifest() -> dict:
+22 -10
View File
@@ -1,6 +1,7 @@
from pathlib import Path from pathlib import Path
import argparse import argparse
from serato_doctor.config import ScanConfig
from serato_doctor.crate_parser import parse_crates from serato_doctor.crate_parser import parse_crates
from serato_doctor.models.library import Library from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_audio from serato_doctor.scanner import scan_audio
@@ -13,28 +14,39 @@ def main():
parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox")) parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox"))
parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv")) parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv"))
parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt")) parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt"))
parser.add_argument(
"--reference-root",
action="append",
default=[],
help="Old library root stored in crates; may be supplied more than once",
)
args = parser.parse_args() args = parser.parse_args()
serato = Path(args.serato) config = ScanConfig.build(
music = Path(args.music) serato=Path(args.serato),
out = Path(args.out) music=Path(args.music),
report = Path(args.report) out=Path(args.out),
report=Path(args.report),
reference_roots=(Path(root) for root in args.reference_root),
)
library = Library.build( library = Library.build(
references=parse_crates(serato / "Subcrates"), references=parse_crates(
tracks=scan_audio(music), config.serato / "Subcrates", config.reference_roots
),
tracks=scan_audio(config.music),
) )
results = library.reconcile_by_filename() results = library.reconcile_by_filename()
missing_count = sum(1 for result in results if not result.exists_by_filename) missing_count = sum(1 for result in results if not result.exists_by_filename)
write_csv(results, out) write_csv(results, config.out)
write_missing_report(results, report) write_missing_report(results, config.report)
print(f"Crate references: {len(library.references)}") print(f"Crate references: {len(library.references)}")
print(f"Disk tracks: {len(library.tracks)}") print(f"Disk tracks: {len(library.tracks)}")
print(f"Missing by filename: {missing_count}") print(f"Missing by filename: {missing_count}")
print(f"CSV: {out}") print(f"CSV: {config.out}")
print(f"Report: {report}") print(f"Report: {config.report}")
if __name__ == "__main__": if __name__ == "__main__":
+25
View File
@@ -0,0 +1,25 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Tuple
@dataclass(frozen=True)
class ScanConfig:
"""Read-only paths used for one library scan."""
serato: Path
music: Path
out: Path
report: Path
reference_roots: Tuple[Path, ...] = ()
@classmethod
def build(
cls,
serato: Path,
music: Path,
out: Path,
report: Path,
reference_roots: Iterable[Path] = (),
) -> "ScanConfig":
return cls(serato, music, out, report, tuple(reference_roots))
+40 -21
View File
@@ -1,4 +1,5 @@
from pathlib import Path from pathlib import Path
from typing import Iterable, Tuple
from serato_doctor.models.crate import Crate from serato_doctor.models.crate import Crate
from serato_doctor.models.reference import TrackReference from serato_doctor.models.reference import TrackReference
@@ -21,45 +22,63 @@ def clean_path(raw: str) -> str:
return raw.strip() return raw.strip()
def load_crate(crate_path: Path) -> Crate: DEFAULT_PATH_MARKERS = ("Users/", "Volumes/")
def path_markers(reference_roots: Iterable[Path]) -> Tuple[str, ...]:
configured = tuple(
root.expanduser().as_posix().strip("/") + "/" for root in reference_roots
)
return configured or DEFAULT_PATH_MARKERS
def load_crate(crate_path: Path, reference_roots: Iterable[Path] = ()) -> Crate:
text = read_crate_text(crate_path) text = read_crate_text(crate_path)
refs = [] refs = []
marker = "Users/djsplice/OneDrive/Jukebox/"
# Serato record markers seen after paths in UTF-16-LE decoded crate data. # Serato record markers seen after paths in UTF-16-LE decoded crate data.
stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"] stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"]
for part in text.split(marker)[1:]: for marker in path_markers(reference_roots):
candidate = marker + part for part in text.split(marker)[1:]:
candidate = marker + part
stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1] stops = [
if not stops: candidate.find(stop)
continue for stop in stop_markers
if candidate.find(stop) != -1
]
if not stops:
continue
raw_path = "/" + candidate[: min(stops)] raw_path = "/" + candidate[: min(stops)]
raw_path = clean_path(raw_path) raw_path = clean_path(raw_path)
path = Path(raw_path) path = Path(raw_path)
refs.append( refs.append(
TrackReference( TrackReference(
source=crate_path, source=crate_path,
path=path, path=path,
filename=path.name, filename=path.name,
)
) )
)
return Crate(path=crate_path, references=tuple(refs)) return Crate(path=crate_path, references=tuple(refs))
def parse_crate(crate_path: Path) -> list[TrackReference]: def parse_crate(
crate_path: Path, reference_roots: Iterable[Path] = ()
) -> list[TrackReference]:
"""Parse references from one crate, preserving the prototype API.""" """Parse references from one crate, preserving the prototype API."""
return list(load_crate(crate_path).references) return list(load_crate(crate_path, reference_roots).references)
def parse_crates(root: Path) -> list[TrackReference]: def parse_crates(
root: Path, reference_roots: Iterable[Path] = ()
) -> list[TrackReference]:
refs = [] refs = []
reference_roots = tuple(reference_roots)
for crate in root.rglob("*.crate"): for crate in root.rglob("*.crate"):
refs.extend(parse_crate(crate)) refs.extend(parse_crate(crate, reference_roots))
return refs return refs
+37 -2
View File
@@ -12,8 +12,8 @@ def test_cli_writes_reports_and_prints_counts(tmp_path, monkeypatch, capsys):
music.mkdir() music.mkdir()
crate_text = ( crate_text = (
"Users/djsplice/OneDrive/Jukebox/Found.mp漳牴k" "Users/sample-user/OneDrive/Jukebox/Found.mp漳牴k"
"Users/djsplice/OneDrive/Jukebox/Missing.mp漳牴k" "Users/sample-user/OneDrive/Jukebox/Missing.mp漳牴k"
) )
(subcrates / "Test.crate").write_bytes(crate_text.encode("utf-16-le")) (subcrates / "Test.crate").write_bytes(crate_text.encode("utf-16-le"))
(music / "Found.mp3").write_bytes(b"synthetic audio") (music / "Found.mp3").write_bytes(b"synthetic audio")
@@ -47,3 +47,38 @@ def test_cli_writes_reports_and_prints_counts(tmp_path, monkeypatch, capsys):
rows = list(csv.DictReader(csv_file)) rows = list(csv.DictReader(csv_file))
assert [row["exists_by_filename"] for row in rows] == ["True", "False"] assert [row["exists_by_filename"] for row in rows] == ["True", "False"]
assert "Total missing references: 1" in report_path.read_text(encoding="utf-8") assert "Total missing references: 1" in report_path.read_text(encoding="utf-8")
def test_cli_accepts_old_reference_root(tmp_path, monkeypatch, capsys):
serato = tmp_path / "serato"
subcrates = serato / "Subcrates"
music = tmp_path / "music"
subcrates.mkdir(parents=True)
music.mkdir()
(subcrates / "Test.crate").write_bytes(
"Archive/Jukebox/Found.mp3otrk".encode("utf-16-le")
)
(music / "Found.mp3").write_bytes(b"synthetic audio")
monkeypatch.setattr(
sys,
"argv",
[
"serato-doctor",
"--serato",
str(serato),
"--music",
str(music),
"--out",
str(tmp_path / "scan.csv"),
"--report",
str(tmp_path / "report.txt"),
"--reference-root",
"/Archive/Jukebox",
],
)
main()
output = capsys.readouterr().out
assert "Crate references: 1" in output
assert "Missing by filename: 0" in output
+17
View File
@@ -0,0 +1,17 @@
from pathlib import Path
from serato_doctor.config import ScanConfig
def test_scan_config_freezes_reference_roots():
roots = (Path(root) for root in ["/old/one", "/old/two"])
config = ScanConfig.build(
serato=Path("serato"),
music=Path("music"),
out=Path("scan.csv"),
report=Path("report.txt"),
reference_roots=roots,
)
assert config.reference_roots == (Path("/old/one"), Path("/old/two"))
+43 -4
View File
@@ -2,7 +2,13 @@ from pathlib import Path
import pytest import pytest
from serato_doctor.crate_parser import clean_path, load_crate, parse_crate from serato_doctor.crate_parser import (
clean_path,
load_crate,
parse_crate,
parse_crates,
path_markers,
)
@pytest.mark.parametrize( @pytest.mark.parametrize(
@@ -22,9 +28,9 @@ def test_parse_crate_extracts_references(tmp_path):
crate_path = tmp_path / "House.crate" crate_path = tmp_path / "House.crate"
crate_text = ( crate_text = (
"header" "header"
"Users/djsplice/OneDrive/Jukebox/House/First.mp漳牴k" "Users/sample-user/OneDrive/Jukebox/House/First.mp漳牴k"
"metadata" "metadata"
"Users/djsplice/OneDrive/Jukebox/House/Second.m4愠otrk" "Users/sample-user/OneDrive/Jukebox/House/Second.m4愠otrk"
) )
crate_path.write_bytes(crate_text.encode("utf-16-le")) crate_path.write_bytes(crate_text.encode("utf-16-le"))
@@ -41,7 +47,40 @@ def test_parse_crate_extracts_references(tmp_path):
def test_parse_crate_ignores_record_without_stop_marker(tmp_path): def test_parse_crate_ignores_record_without_stop_marker(tmp_path):
crate_path = Path(tmp_path) / "Incomplete.crate" crate_path = Path(tmp_path) / "Incomplete.crate"
crate_path.write_bytes( crate_path.write_bytes(
"Users/djsplice/OneDrive/Jukebox/House/Incomplete.mp3".encode("utf-16-le") "Users/sample-user/OneDrive/Jukebox/House/Incomplete.mp3".encode(
"utf-16-le"
)
) )
assert parse_crate(crate_path) == [] assert parse_crate(crate_path) == []
def test_parse_crate_uses_configured_reference_root(tmp_path):
crate_path = tmp_path / "Migrated.crate"
crate_path.write_bytes(
"Archive/Old Library/House/Track.mp3otrk".encode("utf-16-le")
)
references = parse_crate(crate_path, [Path("/Archive/Old Library")])
assert references[0].path == Path("/Archive/Old Library/House/Track.mp3")
def test_default_path_markers_do_not_contain_a_username():
assert path_markers([]) == ("Users/", "Volumes/")
def test_parse_crates_reuses_configured_roots_for_every_crate(tmp_path):
for name in ("First", "Second"):
(tmp_path / f"{name}.crate").write_bytes(
f"Archive/Jukebox/{name}.mp3otrk".encode("utf-16-le")
)
references = parse_crates(
tmp_path, (root for root in [Path("/Archive/Jukebox")])
)
assert {reference.filename for reference in references} == {
"First.mp3",
"Second.mp3",
}