From 4ad5b8a5743f4685e3430bc64b8c14d3e559d7e4 Mon Sep 17 00:00:00 2001 From: Philip Guzman Date: Mon, 29 Jun 2026 23:09:52 -0700 Subject: [PATCH 1/2] Add working crate parser and filesystem scanner --- serato_doctor/cli.py | 46 ++++++++++++++++++++++++++++++++++ serato_doctor/crate_parser.py | 47 +++++++++++++++++++++++++++++++++++ serato_doctor/models.py | 17 +++++++++++++ serato_doctor/scanner.py | 31 +++++++++++++++++++++++ 4 files changed, 141 insertions(+) create mode 100644 serato_doctor/models.py diff --git a/serato_doctor/cli.py b/serato_doctor/cli.py index e69de29..8ba0026 100644 --- a/serato_doctor/cli.py +++ b/serato_doctor/cli.py @@ -0,0 +1,46 @@ +from pathlib import Path +import argparse +import csv + +from serato_doctor.crate_parser import parse_crates +from serato_doctor.scanner import scan_audio + + +def main(): + parser = argparse.ArgumentParser(prog="serato-doctor") + parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_")) + parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox")) + parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv")) + args = parser.parse_args() + + serato = Path(args.serato) + music = Path(args.music) + out = Path(args.out) + + refs = parse_crates(serato / "Subcrates") + disk = scan_audio(music) + + disk_names = {t.filename for t in disk} + + rows = [] + for ref in refs: + rows.append({ + "crate": str(ref.source), + "serato_path": str(ref.path), + "filename": ref.filename, + "exists_by_filename": ref.filename in disk_names, + }) + + with out.open("w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=["crate", "serato_path", "filename", "exists_by_filename"]) + writer.writeheader() + writer.writerows(rows) + + print(f"Crate references: {len(refs)}") + print(f"Disk tracks: {len(disk)}") + print(f"Missing by filename: {sum(1 for r in rows if not r['exists_by_filename'])}") + print(f"Wrote: {out}") + + +if __name__ == "__main__": + main() diff --git a/serato_doctor/crate_parser.py b/serato_doctor/crate_parser.py index e69de29..947a538 100644 --- a/serato_doctor/crate_parser.py +++ b/serato_doctor/crate_parser.py @@ -0,0 +1,47 @@ +from pathlib import Path +import re + +from serato_doctor.models import TrackReference + +AUDIO_EXTS = "mp3|m4a|wav|aif|aiff|flac|MP3|M4A|WAV|AIF|AIFF|FLAC" + + +def read_crate_text(crate_path: Path) -> str: + raw = crate_path.read_bytes() + for enc in ("utf-16-be", "utf-16-le", "utf-8", "latin1"): + text = raw.decode(enc, errors="ignore") + if "Users" in text or "Jukebox" in text: + return text + return raw.decode("latin1", errors="ignore") + + +def parse_crate(crate_path: Path) -> list[TrackReference]: + text = read_crate_text(crate_path) + + # Serato crate files often decode with weird spacing/null-ish characters. + # This finds paths from /Users/... through the audio extension without + # greedily scanning the entire file. + pattern = rf"/?Users/[^\r\n]+?\.(?:{AUDIO_EXTS})" + + refs = [] + for match in re.finditer(pattern, text): + raw_path = "/" + match.group(0).lstrip("/") + raw_path = raw_path.replace("\x00", "") + path = Path(raw_path) + + refs.append( + TrackReference( + source=crate_path, + path=path, + filename=path.name, + ) + ) + + return refs + + +def parse_crates(root: Path) -> list[TrackReference]: + refs = [] + for crate in root.rglob("*.crate"): + refs.extend(parse_crate(crate)) + return refs diff --git a/serato_doctor/models.py b/serato_doctor/models.py new file mode 100644 index 0000000..22466b9 --- /dev/null +++ b/serato_doctor/models.py @@ -0,0 +1,17 @@ +from dataclasses import dataclass +from pathlib import Path + + +@dataclass(frozen=True) +class TrackReference: + source: Path + path: Path + filename: str + + +@dataclass(frozen=True) +class DiskTrack: + path: Path + filename: str + size: int + suffix: str diff --git a/serato_doctor/scanner.py b/serato_doctor/scanner.py index e69de29..d12ee0e 100644 --- a/serato_doctor/scanner.py +++ b/serato_doctor/scanner.py @@ -0,0 +1,31 @@ +from pathlib import Path + +from serato_doctor.models import DiskTrack + +AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"} + + +def scan_audio(folder: Path) -> list[DiskTrack]: + tracks = [] + + for path in folder.rglob("*"): + if not path.is_file(): + continue + if path.suffix.lower() not in AUDIO_SUFFIXES: + continue + + try: + stat = path.stat() + except OSError: + continue + + tracks.append( + DiskTrack( + path=path, + filename=path.name, + size=stat.st_size, + suffix=path.suffix.lower(), + ) + ) + + return tracks From 826428f3808450c382d7a70e2db6000110db702c Mon Sep 17 00:00:00 2001 From: Philip Guzman Date: Tue, 30 Jun 2026 08:43:11 -0700 Subject: [PATCH 2/2] Improve Serato crate parser for UTF-16 LE path records --- .gitignore | 3 +++ serato_doctor/crate_parser.py | 49 +++++++++++++++++++++-------------- 2 files changed, 33 insertions(+), 19 deletions(-) diff --git a/.gitignore b/.gitignore index c146d88..b17dc23 100644 --- a/.gitignore +++ b/.gitignore @@ -11,3 +11,6 @@ reports/ # Never commit personal Serato data database V2 *.crate + +# macOS +.DS_Store diff --git a/serato_doctor/crate_parser.py b/serato_doctor/crate_parser.py index 947a538..6062380 100644 --- a/serato_doctor/crate_parser.py +++ b/serato_doctor/crate_parser.py @@ -1,34 +1,45 @@ from pathlib import Path -import re from serato_doctor.models import TrackReference -AUDIO_EXTS = "mp3|m4a|wav|aif|aiff|flac|MP3|M4A|WAV|AIF|AIFF|FLAC" - def read_crate_text(crate_path: Path) -> str: - raw = crate_path.read_bytes() - for enc in ("utf-16-be", "utf-16-le", "utf-8", "latin1"): - text = raw.decode(enc, errors="ignore") - if "Users" in text or "Jukebox" in text: - return text - return raw.decode("latin1", errors="ignore") + return crate_path.read_bytes().decode("utf-16-le", errors="ignore").replace("\x00", "") + + +def clean_path(raw: str) -> str: + # Common Serato decode artifacts where final extension char gets merged. + raw = raw.replace(".mp漳", ".mp3") + raw = raw.replace(".MP漳", ".MP3") + raw = raw.replace(".m4愠", ".m4a") + raw = raw.replace(".M4愠", ".M4A") + raw = raw.replace(".wa瘠", ".wav") + raw = raw.replace(".WA瘠", ".WAV") + raw = raw.replace(".ai映", ".aif") + raw = raw.replace(".AI映", ".AIF") + return raw.strip() def parse_crate(crate_path: Path) -> list[TrackReference]: text = read_crate_text(crate_path) - - # Serato crate files often decode with weird spacing/null-ish characters. - # This finds paths from /Users/... through the audio extension without - # greedily scanning the entire file. - pattern = rf"/?Users/[^\r\n]+?\.(?:{AUDIO_EXTS})" - refs = [] - for match in re.finditer(pattern, text): - raw_path = "/" + match.group(0).lstrip("/") - raw_path = raw_path.replace("\x00", "") - path = Path(raw_path) + marker = "Users/djsplice/OneDrive/Jukebox/" + + # Serato record markers seen after paths in UTF-16-LE decoded crate data. + stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"] + + for part in text.split(marker)[1:]: + candidate = marker + part + + stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1] + if not stops: + continue + + raw_path = "/" + candidate[: min(stops)] + raw_path = clean_path(raw_path) + + path = Path(raw_path) refs.append( TrackReference( source=crate_path,