diff --git a/.gitignore b/.gitignore index c146d88..b17dc23 100644 --- a/.gitignore +++ b/.gitignore @@ -11,3 +11,6 @@ reports/ # Never commit personal Serato data database V2 *.crate + +# macOS +.DS_Store diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..0d25061 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,13 @@ +# Contributing + +## Branching + +- `main` is stable. +- `develop` is the integration branch. +- Feature branches use: `feature/`. + +## Safety Rules + +Never commit personal music library files, Serato databases, or crates. + +Never write repair code without dry-run mode, backup plan, and rollback log. diff --git a/ROADMAP.md b/ROADMAP.md new file mode 100644 index 0000000..ef2fd8d --- /dev/null +++ b/ROADMAP.md @@ -0,0 +1,48 @@ +# Serato Doctor Roadmap + +## v0.1 — Library Inspector + +- [x] Project repository +- [x] Filesystem scanner +- [x] Serato crate parser +- [x] Missing reference CSV report +- [x] Grouped missing reference report +- [ ] HTML health dashboard +- [ ] Test suite +- [ ] Sample library fixtures +- [ ] Database V2 read-only parser + +## v0.2 — Diagnostics + +- [ ] Duplicate filename detection +- [ ] Duplicate audio hash detection +- [ ] Broken symlink detection +- [ ] Orphaned audio detection +- [ ] OneDrive rename detection +- [ ] Crate classification: static vs smart/dynamic +- [ ] Library health score + +## v0.3 — Safe Repair + +- [ ] Dry-run repair plan +- [ ] Backup before repair +- [ ] Compatibility symlink creation +- [ ] Compatibility copy creation +- [ ] Rename repair +- [ ] Rollback log + +## v0.4 — Migration Wizard + +- [ ] Move library root +- [ ] Cloud provider migration +- [ ] External drive migration +- [ ] Verify moved library +- [ ] Update application references + +## v1.0 — DJ Library Doctor + +- [ ] Desktop UI +- [ ] Serato support +- [ ] Rekordbox support +- [ ] VirtualDJ support +- [ ] Engine DJ support diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..a7b2e11 --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,11 @@ +# Architecture + +Serato Doctor is designed as a DJ library inspection, repair, and migration platform. + +## Design Principles + +1. Read-only by default. +2. Every repair must support preview/dry-run. +3. Every repair must create a backup or rollback path. +4. Application-specific logic lives in engines. +5. Core matching and scanning logic should be application-agnostic. diff --git a/docs/case-studies/onedrive-mac-migration.md b/docs/case-studies/onedrive-mac-migration.md new file mode 100644 index 0000000..86c29d6 --- /dev/null +++ b/docs/case-studies/onedrive-mac-migration.md @@ -0,0 +1,26 @@ +# Case Study: OneDrive Mac Migration + +## Scenario + +A large Serato DJ library was migrated from an older Mac to a newer Mac using OneDrive. + +## Symptoms + +- OneDrive client stuck syncing +- Duplicate OneDrive folders +- Thousands of files renamed with trailing ` 2` +- Serato reported many tracks as missing +- Some files existed on disk but still appeared orange in Serato + +## Findings + +- OneDrive sync state was rebuilt successfully +- Thousands of orphaned filename conflicts were repaired +- Some Serato references were stale database objects, not missing files +- Smart/dynamic crates should be classified separately from static user crates + +## Lessons + +- Filesystem health and Serato database health are separate problems +- Smart crates should not be treated the same as static crates +- Repair tools must be read-only by default and generate a plan before changing anything diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..dbd025f --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,9 @@ +[project] +name = "serato-doctor" +version = "0.1.0" +description = "Inspect, diagnose, repair, and migrate DJ libraries." +requires-python = ">=3.9" +dependencies = [] + +[project.scripts] +serato-doctor = "serato_doctor.cli:main" diff --git a/serato_doctor/cli.py b/serato_doctor/cli.py index e69de29..1df05c5 100644 --- a/serato_doctor/cli.py +++ b/serato_doctor/cli.py @@ -0,0 +1,41 @@ +from pathlib import Path +import argparse + +from serato_doctor.crate_parser import parse_crates +from serato_doctor.models.library import Library +from serato_doctor.scanner import scan_audio +from serato_doctor.report import write_csv, write_missing_report + + +def main(): + parser = argparse.ArgumentParser(prog="serato-doctor") + parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_")) + parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox")) + parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv")) + parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt")) + args = parser.parse_args() + + serato = Path(args.serato) + music = Path(args.music) + out = Path(args.out) + report = Path(args.report) + + library = Library.build( + references=parse_crates(serato / "Subcrates"), + tracks=scan_audio(music), + ) + results = library.reconcile_by_filename() + missing_count = sum(1 for result in results if not result.exists_by_filename) + + write_csv(results, out) + write_missing_report(results, report) + + print(f"Crate references: {len(library.references)}") + print(f"Disk tracks: {len(library.tracks)}") + print(f"Missing by filename: {missing_count}") + print(f"CSV: {out}") + print(f"Report: {report}") + + +if __name__ == "__main__": + main() diff --git a/serato_doctor/crate_parser.py b/serato_doctor/crate_parser.py index e69de29..b53e23f 100644 --- a/serato_doctor/crate_parser.py +++ b/serato_doctor/crate_parser.py @@ -0,0 +1,65 @@ +from pathlib import Path + +from serato_doctor.models.crate import Crate +from serato_doctor.models.reference import TrackReference + + +def read_crate_text(crate_path: Path) -> str: + return crate_path.read_bytes().decode("utf-16-le", errors="ignore").replace("\x00", "") + + +def clean_path(raw: str) -> str: + # Common Serato decode artifacts where final extension char gets merged. + raw = raw.replace(".mp漳", ".mp3") + raw = raw.replace(".MP漳", ".MP3") + raw = raw.replace(".m4愠", ".m4a") + raw = raw.replace(".M4愠", ".M4A") + raw = raw.replace(".wa瘠", ".wav") + raw = raw.replace(".WA瘠", ".WAV") + raw = raw.replace(".ai映", ".aif") + raw = raw.replace(".AI映", ".AIF") + return raw.strip() + + +def load_crate(crate_path: Path) -> Crate: + text = read_crate_text(crate_path) + refs = [] + + marker = "Users/djsplice/OneDrive/Jukebox/" + + # Serato record markers seen after paths in UTF-16-LE decoded crate data. + stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"] + + for part in text.split(marker)[1:]: + candidate = marker + part + + stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1] + if not stops: + continue + + raw_path = "/" + candidate[: min(stops)] + raw_path = clean_path(raw_path) + + path = Path(raw_path) + refs.append( + TrackReference( + source=crate_path, + path=path, + filename=path.name, + ) + ) + + return Crate(path=crate_path, references=tuple(refs)) + + +def parse_crate(crate_path: Path) -> list[TrackReference]: + """Parse references from one crate, preserving the prototype API.""" + + return list(load_crate(crate_path).references) + + +def parse_crates(root: Path) -> list[TrackReference]: + refs = [] + for crate in root.rglob("*.crate"): + refs.extend(parse_crate(crate)) + return refs diff --git a/serato_doctor/models/__init__.py b/serato_doctor/models/__init__.py new file mode 100644 index 0000000..54dc054 --- /dev/null +++ b/serato_doctor/models/__init__.py @@ -0,0 +1,6 @@ +from serato_doctor.models.crate import Crate +from serato_doctor.models.library import Library +from serato_doctor.models.reference import ReferenceResult, TrackReference +from serato_doctor.models.track import DiskTrack + +__all__ = ["Crate", "DiskTrack", "Library", "ReferenceResult", "TrackReference"] diff --git a/serato_doctor/models/crate.py b/serato_doctor/models/crate.py new file mode 100644 index 0000000..dc588e9 --- /dev/null +++ b/serato_doctor/models/crate.py @@ -0,0 +1,13 @@ +from dataclasses import dataclass +from pathlib import Path +from typing import Tuple + +from serato_doctor.models.reference import TrackReference + + +@dataclass(frozen=True) +class Crate: + """A Serato crate and the track references parsed from it.""" + + path: Path + references: Tuple[TrackReference, ...] diff --git a/serato_doctor/models/library.py b/serato_doctor/models/library.py new file mode 100644 index 0000000..e099dff --- /dev/null +++ b/serato_doctor/models/library.py @@ -0,0 +1,31 @@ +from dataclasses import dataclass +from typing import Iterable, Tuple + +from serato_doctor.models.reference import ReferenceResult, TrackReference +from serato_doctor.models.track import DiskTrack + + +@dataclass(frozen=True) +class Library: + """The read-only view of crate references and audio found on disk.""" + + references: Tuple[TrackReference, ...] + tracks: Tuple[DiskTrack, ...] + + @classmethod + def build( + cls, + references: Iterable[TrackReference], + tracks: Iterable[DiskTrack], + ) -> "Library": + return cls(tuple(references), tuple(tracks)) + + def reconcile_by_filename(self) -> Tuple[ReferenceResult, ...]: + disk_names = {track.filename for track in self.tracks} + return tuple( + ReferenceResult( + reference=reference, + exists_by_filename=reference.filename in disk_names, + ) + for reference in self.references + ) diff --git a/serato_doctor/models/reference.py b/serato_doctor/models/reference.py new file mode 100644 index 0000000..522bf4e --- /dev/null +++ b/serato_doctor/models/reference.py @@ -0,0 +1,27 @@ +from dataclasses import dataclass +from pathlib import Path + + +@dataclass(frozen=True) +class TrackReference: + """A track path referenced by a Serato crate.""" + + source: Path + path: Path + filename: str + + +@dataclass(frozen=True) +class ReferenceResult: + """The filename-level reconciliation result for a crate reference.""" + + reference: TrackReference + exists_by_filename: bool + + def as_row(self) -> dict: + return { + "crate": str(self.reference.source), + "serato_path": str(self.reference.path), + "filename": self.reference.filename, + "exists_by_filename": self.exists_by_filename, + } diff --git a/serato_doctor/models/track.py b/serato_doctor/models/track.py new file mode 100644 index 0000000..6ce7856 --- /dev/null +++ b/serato_doctor/models/track.py @@ -0,0 +1,12 @@ +from dataclasses import dataclass +from pathlib import Path + + +@dataclass(frozen=True) +class DiskTrack: + """An audio file discovered on disk.""" + + path: Path + filename: str + size: int + suffix: str diff --git a/serato_doctor/report.py b/serato_doctor/report.py index e69de29..8adc7ff 100644 --- a/serato_doctor/report.py +++ b/serato_doctor/report.py @@ -0,0 +1,47 @@ +from collections import Counter, defaultdict +from pathlib import Path +from typing import Iterable +import csv + +from serato_doctor.models.reference import ReferenceResult + + +def write_missing_report(results: Iterable[ReferenceResult], out: Path) -> None: + rows = [result.as_row() for result in results] + missing = [r for r in rows if not r["exists_by_filename"]] + + crate_counts = Counter(r["crate"] for r in missing) + filename_counts = Counter(r["filename"] for r in missing) + + with out.open("w", encoding="utf-8") as f: + f.write("# Serato Doctor Missing Report\n\n") + f.write(f"Total missing references: {len(missing)}\n\n") + + f.write("## Missing by crate\n\n") + for crate, count in crate_counts.most_common(): + f.write(f"{count:5} {crate}\n") + + f.write("\n## Most common missing filenames\n\n") + for filename, count in filename_counts.most_common(100): + f.write(f"{count:5} {filename}\n") + + f.write("\n## Detail\n\n") + by_crate = defaultdict(list) + for r in missing: + by_crate[r["crate"]].append(r["filename"]) + + for crate, names in sorted(by_crate.items()): + f.write(f"\n### {crate}\n") + for name in sorted(set(names)): + f.write(f"- {name}\n") + + +def write_csv(results: Iterable[ReferenceResult], out: Path) -> None: + rows = [result.as_row() for result in results] + with out.open("w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter( + f, + fieldnames=["crate", "serato_path", "filename", "exists_by_filename"], + ) + writer.writeheader() + writer.writerows(rows) diff --git a/serato_doctor/scanner.py b/serato_doctor/scanner.py index e69de29..319aa7e 100644 --- a/serato_doctor/scanner.py +++ b/serato_doctor/scanner.py @@ -0,0 +1,31 @@ +from pathlib import Path + +from serato_doctor.models.track import DiskTrack + +AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"} + + +def scan_audio(folder: Path) -> list[DiskTrack]: + tracks = [] + + for path in folder.rglob("*"): + if not path.is_file(): + continue + if path.suffix.lower() not in AUDIO_SUFFIXES: + continue + + try: + stat = path.stat() + except OSError: + continue + + tracks.append( + DiskTrack( + path=path, + filename=path.name, + size=stat.st_size, + suffix=path.suffix.lower(), + ) + ) + + return tracks