Compare commits

..

6 Commits

Author SHA1 Message Date
Philip Guzman e4d5a32de2 Merge feature/configuration into develop 2026-06-30 17:35:59 -07:00
Philip Guzman ec671079ba Add configurable library reference roots 2026-06-30 17:20:08 -07:00
Philip Guzman 4e85b5912b Merge feature/sample-library into develop 2026-06-30 17:18:06 -07:00
Philip Guzman d4fe61cf8c Merge feature/test-foundation into develop 2026-06-30 17:18:06 -07:00
Philip Guzman c4d08bfe27 Merge feature/library-model into develop 2026-06-30 17:18:06 -07:00
Philip Guzman bb74c14742 Add synthetic sample library 2026-06-30 17:15:56 -07:00
14 changed files with 397 additions and 38 deletions
+1
View File
@@ -4,6 +4,7 @@ __pycache__/
.venv/
*.egg-info/
reports/
samples/small-library/generated/
*.sqlite
*.db
*.csv
+2 -1
View File
@@ -9,8 +9,9 @@
- [x] Grouped missing reference report
- [ ] HTML health dashboard
- [x] Test suite
- [ ] Sample library fixtures
- [x] Sample library fixtures
- [ ] Database V2 read-only parser
- [x] Configuration
## v0.2 — Diagnostics
+25
View File
@@ -0,0 +1,25 @@
# Scan Configuration
## Problem
The prototype recognized crate paths by matching one developer's historical
OneDrive path. That made otherwise valid libraries invisible to the parser.
## Architecture
`ScanConfig` owns the paths for a read-only scan. The CLI accepts repeatable
`--reference-root` options for roots embedded in crates before a migration. The
parser uses those roots as record boundaries. Without explicit roots, it recognizes
generic macOS home and mounted-volume paths without embedding a username.
## Edge Cases
- A library may have references from more than one historical root.
- Root paths may contain spaces or have leading/trailing separators.
- Existing `/Users/...` and `/Volumes/...` crates must work without new flags.
- A record without a recognized terminator remains ignored.
## Verification
Tests cover default root discovery, a custom migrated root, the CLI option, and
the existing synthetic sample library. The scanner remains read-only.
+26
View File
@@ -0,0 +1,26 @@
# Synthetic Sample Library
## Problem
Real Serato libraries contain private paths, listening history, and copyrighted
music. Contributors still need a repeatable library for tests and demonstrations.
## Architecture
`samples/small-library/manifest.json` describes a compact migration scenario.
`generate.py` turns that manifest into fake audio files and UTF-16-LE crate files
inside an ignored `generated` directory. The generated files are disposable and
contain no real audio or user library data.
## Edge Cases
- Five manually maintained crates and two smart/dynamic crates.
- Mixed supported audio extensions.
- One reference whose file is absent.
- One stale reference whose file has been renamed.
- References shared by static and smart crates.
## Verification
`pytest` generates the sample in a temporary directory and verifies its crate,
track, and missing-reference counts through the production parser and scanner.
+25
View File
@@ -0,0 +1,25 @@
# Small Synthetic Library
This fixture models a small Serato migration without containing music or personal
library data. It has 10 fake audio files, five static crates, two smart crates,
one absent song, and one renamed song.
Generate it with:
```shell
python3 samples/small-library/generate.py
```
Analyze it with:
```shell
python3 -m serato_doctor.cli \
--serato samples/small-library/generated/Serato/_Serato_ \
--music samples/small-library/generated/Music \
--out samples/small-library/generated/scan.csv \
--report samples/small-library/generated/report.txt
```
Expected CLI counts are 13 crate references, 10 disk tracks, and 2 references
missing by filename. Everything under `generated/` is disposable and ignored by
Git.
+50
View File
@@ -0,0 +1,50 @@
"""Generate a disposable, synthetic Serato library from the sample manifest."""
import argparse
import json
from pathlib import Path
from typing import Optional
SAMPLE_ROOT = Path(__file__).parent
SERATO_PATH_PREFIX = "Users/sample-user/OneDrive/Jukebox/"
def load_manifest() -> dict:
return json.loads((SAMPLE_ROOT / "manifest.json").read_text(encoding="utf-8"))
def build_sample(output: Optional[Path] = None) -> Path:
root = output or SAMPLE_ROOT / "generated"
manifest = load_manifest()
music_root = root / "Music"
crate_root = root / "Serato" / "_Serato_" / "Subcrates"
crate_root.mkdir(parents=True, exist_ok=True)
for relative_path in manifest["tracks"]:
track_path = music_root / relative_path
track_path.parent.mkdir(parents=True, exist_ok=True)
track_path.write_bytes(
f"Synthetic Serato Doctor fixture: {relative_path}\n".encode("utf-8")
)
for crate in manifest["crates"]:
records = "".join(
f"{SERATO_PATH_PREFIX}{relative_path}otrk"
for relative_path in crate["references"]
)
(crate_root / crate["name"]).write_bytes(records.encode("utf-16-le"))
return root
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--output", type=Path)
args = parser.parse_args()
root = build_sample(args.output)
print(f"Generated synthetic library: {root}")
if __name__ == "__main__":
main()
+56
View File
@@ -0,0 +1,56 @@
{
"tracks": [
"House/First.mp3",
"House/Second.m4a",
"Open Format/Third.wav",
"Open Format/Fourth.aif",
"Classics/Fifth.mp3",
"Classics/Sixth.flac",
"Warmup/Seventh.mp3",
"Warmup/Eighth.mp3",
"Renamed/New Name.mp3",
"Bonus/Ninth.MP3"
],
"crates": [
{
"name": "House.crate",
"type": "static",
"references": ["House/First.mp3", "House/Second.m4a", "House/Missing.mp3"]
},
{
"name": "Open Format.crate",
"type": "static",
"references": ["Open Format/Third.wav", "Open Format/Fourth.aif"]
},
{
"name": "Classics.crate",
"type": "static",
"references": ["Classics/Fifth.mp3", "Classics/Sixth.flac"]
},
{
"name": "Renamed Tracks.crate",
"type": "static",
"references": ["Renamed/Old Name.mp3"]
},
{
"name": "Bonus.crate",
"type": "static",
"references": ["Bonus/Ninth.MP3"]
},
{
"name": "Compatible by key.crate",
"type": "smart",
"references": ["House/First.mp3", "House/Second.m4a"]
},
{
"name": "Smart Warmup.crate",
"type": "smart",
"references": ["Warmup/Seventh.mp3", "Warmup/Eighth.mp3"]
}
],
"expected": {
"disk_tracks": 10,
"crate_references": 13,
"missing_by_filename": ["Missing.mp3", "Old Name.mp3"]
}
}
+22 -10
View File
@@ -1,6 +1,7 @@
from pathlib import Path
import argparse
from serato_doctor.config import ScanConfig
from serato_doctor.crate_parser import parse_crates
from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_audio
@@ -13,28 +14,39 @@ def main():
parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox"))
parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv"))
parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt"))
parser.add_argument(
"--reference-root",
action="append",
default=[],
help="Old library root stored in crates; may be supplied more than once",
)
args = parser.parse_args()
serato = Path(args.serato)
music = Path(args.music)
out = Path(args.out)
report = Path(args.report)
config = ScanConfig.build(
serato=Path(args.serato),
music=Path(args.music),
out=Path(args.out),
report=Path(args.report),
reference_roots=(Path(root) for root in args.reference_root),
)
library = Library.build(
references=parse_crates(serato / "Subcrates"),
tracks=scan_audio(music),
references=parse_crates(
config.serato / "Subcrates", config.reference_roots
),
tracks=scan_audio(config.music),
)
results = library.reconcile_by_filename()
missing_count = sum(1 for result in results if not result.exists_by_filename)
write_csv(results, out)
write_missing_report(results, report)
write_csv(results, config.out)
write_missing_report(results, config.report)
print(f"Crate references: {len(library.references)}")
print(f"Disk tracks: {len(library.tracks)}")
print(f"Missing by filename: {missing_count}")
print(f"CSV: {out}")
print(f"Report: {report}")
print(f"CSV: {config.out}")
print(f"Report: {config.report}")
if __name__ == "__main__":
+25
View File
@@ -0,0 +1,25 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Tuple
@dataclass(frozen=True)
class ScanConfig:
"""Read-only paths used for one library scan."""
serato: Path
music: Path
out: Path
report: Path
reference_roots: Tuple[Path, ...] = ()
@classmethod
def build(
cls,
serato: Path,
music: Path,
out: Path,
report: Path,
reference_roots: Iterable[Path] = (),
) -> "ScanConfig":
return cls(serato, music, out, report, tuple(reference_roots))
+40 -21
View File
@@ -1,4 +1,5 @@
from pathlib import Path
from typing import Iterable, Tuple
from serato_doctor.models.crate import Crate
from serato_doctor.models.reference import TrackReference
@@ -21,45 +22,63 @@ def clean_path(raw: str) -> str:
return raw.strip()
def load_crate(crate_path: Path) -> Crate:
DEFAULT_PATH_MARKERS = ("Users/", "Volumes/")
def path_markers(reference_roots: Iterable[Path]) -> Tuple[str, ...]:
configured = tuple(
root.expanduser().as_posix().strip("/") + "/" for root in reference_roots
)
return configured or DEFAULT_PATH_MARKERS
def load_crate(crate_path: Path, reference_roots: Iterable[Path] = ()) -> Crate:
text = read_crate_text(crate_path)
refs = []
marker = "Users/djsplice/OneDrive/Jukebox/"
# Serato record markers seen after paths in UTF-16-LE decoded crate data.
stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"]
for part in text.split(marker)[1:]:
candidate = marker + part
for marker in path_markers(reference_roots):
for part in text.split(marker)[1:]:
candidate = marker + part
stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1]
if not stops:
continue
stops = [
candidate.find(stop)
for stop in stop_markers
if candidate.find(stop) != -1
]
if not stops:
continue
raw_path = "/" + candidate[: min(stops)]
raw_path = clean_path(raw_path)
raw_path = "/" + candidate[: min(stops)]
raw_path = clean_path(raw_path)
path = Path(raw_path)
refs.append(
TrackReference(
source=crate_path,
path=path,
filename=path.name,
path = Path(raw_path)
refs.append(
TrackReference(
source=crate_path,
path=path,
filename=path.name,
)
)
)
return Crate(path=crate_path, references=tuple(refs))
def parse_crate(crate_path: Path) -> list[TrackReference]:
def parse_crate(
crate_path: Path, reference_roots: Iterable[Path] = ()
) -> list[TrackReference]:
"""Parse references from one crate, preserving the prototype API."""
return list(load_crate(crate_path).references)
return list(load_crate(crate_path, reference_roots).references)
def parse_crates(root: Path) -> list[TrackReference]:
def parse_crates(
root: Path, reference_roots: Iterable[Path] = ()
) -> list[TrackReference]:
refs = []
reference_roots = tuple(reference_roots)
for crate in root.rglob("*.crate"):
refs.extend(parse_crate(crate))
refs.extend(parse_crate(crate, reference_roots))
return refs
+37 -2
View File
@@ -12,8 +12,8 @@ def test_cli_writes_reports_and_prints_counts(tmp_path, monkeypatch, capsys):
music.mkdir()
crate_text = (
"Users/djsplice/OneDrive/Jukebox/Found.mp漳牴k"
"Users/djsplice/OneDrive/Jukebox/Missing.mp漳牴k"
"Users/sample-user/OneDrive/Jukebox/Found.mp漳牴k"
"Users/sample-user/OneDrive/Jukebox/Missing.mp漳牴k"
)
(subcrates / "Test.crate").write_bytes(crate_text.encode("utf-16-le"))
(music / "Found.mp3").write_bytes(b"synthetic audio")
@@ -47,3 +47,38 @@ def test_cli_writes_reports_and_prints_counts(tmp_path, monkeypatch, capsys):
rows = list(csv.DictReader(csv_file))
assert [row["exists_by_filename"] for row in rows] == ["True", "False"]
assert "Total missing references: 1" in report_path.read_text(encoding="utf-8")
def test_cli_accepts_old_reference_root(tmp_path, monkeypatch, capsys):
serato = tmp_path / "serato"
subcrates = serato / "Subcrates"
music = tmp_path / "music"
subcrates.mkdir(parents=True)
music.mkdir()
(subcrates / "Test.crate").write_bytes(
"Archive/Jukebox/Found.mp3otrk".encode("utf-16-le")
)
(music / "Found.mp3").write_bytes(b"synthetic audio")
monkeypatch.setattr(
sys,
"argv",
[
"serato-doctor",
"--serato",
str(serato),
"--music",
str(music),
"--out",
str(tmp_path / "scan.csv"),
"--report",
str(tmp_path / "report.txt"),
"--reference-root",
"/Archive/Jukebox",
],
)
main()
output = capsys.readouterr().out
assert "Crate references: 1" in output
assert "Missing by filename: 0" in output
+17
View File
@@ -0,0 +1,17 @@
from pathlib import Path
from serato_doctor.config import ScanConfig
def test_scan_config_freezes_reference_roots():
roots = (Path(root) for root in ["/old/one", "/old/two"])
config = ScanConfig.build(
serato=Path("serato"),
music=Path("music"),
out=Path("scan.csv"),
report=Path("report.txt"),
reference_roots=roots,
)
assert config.reference_roots == (Path("/old/one"), Path("/old/two"))
+43 -4
View File
@@ -2,7 +2,13 @@ from pathlib import Path
import pytest
from serato_doctor.crate_parser import clean_path, load_crate, parse_crate
from serato_doctor.crate_parser import (
clean_path,
load_crate,
parse_crate,
parse_crates,
path_markers,
)
@pytest.mark.parametrize(
@@ -22,9 +28,9 @@ def test_parse_crate_extracts_references(tmp_path):
crate_path = tmp_path / "House.crate"
crate_text = (
"header"
"Users/djsplice/OneDrive/Jukebox/House/First.mp漳牴k"
"Users/sample-user/OneDrive/Jukebox/House/First.mp漳牴k"
"metadata"
"Users/djsplice/OneDrive/Jukebox/House/Second.m4愠otrk"
"Users/sample-user/OneDrive/Jukebox/House/Second.m4愠otrk"
)
crate_path.write_bytes(crate_text.encode("utf-16-le"))
@@ -41,7 +47,40 @@ def test_parse_crate_extracts_references(tmp_path):
def test_parse_crate_ignores_record_without_stop_marker(tmp_path):
crate_path = Path(tmp_path) / "Incomplete.crate"
crate_path.write_bytes(
"Users/djsplice/OneDrive/Jukebox/House/Incomplete.mp3".encode("utf-16-le")
"Users/sample-user/OneDrive/Jukebox/House/Incomplete.mp3".encode(
"utf-16-le"
)
)
assert parse_crate(crate_path) == []
def test_parse_crate_uses_configured_reference_root(tmp_path):
crate_path = tmp_path / "Migrated.crate"
crate_path.write_bytes(
"Archive/Old Library/House/Track.mp3otrk".encode("utf-16-le")
)
references = parse_crate(crate_path, [Path("/Archive/Old Library")])
assert references[0].path == Path("/Archive/Old Library/House/Track.mp3")
def test_default_path_markers_do_not_contain_a_username():
assert path_markers([]) == ("Users/", "Volumes/")
def test_parse_crates_reuses_configured_roots_for_every_crate(tmp_path):
for name in ("First", "Second"):
(tmp_path / f"{name}.crate").write_bytes(
f"Archive/Jukebox/{name}.mp3otrk".encode("utf-16-le")
)
references = parse_crates(
tmp_path, (root for root in [Path("/Archive/Jukebox")])
)
assert {reference.filename for reference in references} == {
"First.mp3",
"Second.mp3",
}
+28
View File
@@ -0,0 +1,28 @@
import runpy
from pathlib import Path
from serato_doctor.crate_parser import parse_crates
from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_audio
def test_generated_sample_library_has_expected_scenario(tmp_path):
generator_path = (
Path(__file__).parents[1] / "samples" / "small-library" / "generate.py"
)
build_sample = runpy.run_path(str(generator_path))["build_sample"]
sample_root = build_sample(tmp_path / "sample")
references = parse_crates(sample_root / "Serato" / "_Serato_" / "Subcrates")
tracks = scan_audio(sample_root / "Music")
results = Library.build(references, tracks).reconcile_by_filename()
missing = {
result.reference.filename
for result in results
if not result.exists_by_filename
}
assert len(list((sample_root / "Serato").rglob("*.crate"))) == 7
assert len(references) == 13
assert len(tracks) == 10
assert missing == {"Missing.mp3", "Old Name.mp3"}