diff --git a/.gitignore b/.gitignore index c146d88..b17dc23 100644 --- a/.gitignore +++ b/.gitignore @@ -11,3 +11,6 @@ reports/ # Never commit personal Serato data database V2 *.crate + +# macOS +.DS_Store diff --git a/serato_doctor/crate_parser.py b/serato_doctor/crate_parser.py index e69de29..6062380 100644 --- a/serato_doctor/crate_parser.py +++ b/serato_doctor/crate_parser.py @@ -0,0 +1,58 @@ +from pathlib import Path + +from serato_doctor.models import TrackReference + + +def read_crate_text(crate_path: Path) -> str: + return crate_path.read_bytes().decode("utf-16-le", errors="ignore").replace("\x00", "") + + +def clean_path(raw: str) -> str: + # Common Serato decode artifacts where final extension char gets merged. + raw = raw.replace(".mp漳", ".mp3") + raw = raw.replace(".MP漳", ".MP3") + raw = raw.replace(".m4愠", ".m4a") + raw = raw.replace(".M4愠", ".M4A") + raw = raw.replace(".wa瘠", ".wav") + raw = raw.replace(".WA瘠", ".WAV") + raw = raw.replace(".ai映", ".aif") + raw = raw.replace(".AI映", ".AIF") + return raw.strip() + + +def parse_crate(crate_path: Path) -> list[TrackReference]: + text = read_crate_text(crate_path) + refs = [] + + marker = "Users/djsplice/OneDrive/Jukebox/" + + # Serato record markers seen after paths in UTF-16-LE decoded crate data. + stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"] + + for part in text.split(marker)[1:]: + candidate = marker + part + + stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1] + if not stops: + continue + + raw_path = "/" + candidate[: min(stops)] + raw_path = clean_path(raw_path) + + path = Path(raw_path) + refs.append( + TrackReference( + source=crate_path, + path=path, + filename=path.name, + ) + ) + + return refs + + +def parse_crates(root: Path) -> list[TrackReference]: + refs = [] + for crate in root.rglob("*.crate"): + refs.extend(parse_crate(crate)) + return refs diff --git a/serato_doctor/models.py b/serato_doctor/models.py new file mode 100644 index 0000000..22466b9 --- /dev/null +++ b/serato_doctor/models.py @@ -0,0 +1,17 @@ +from dataclasses import dataclass +from pathlib import Path + + +@dataclass(frozen=True) +class TrackReference: + source: Path + path: Path + filename: str + + +@dataclass(frozen=True) +class DiskTrack: + path: Path + filename: str + size: int + suffix: str diff --git a/serato_doctor/scanner.py b/serato_doctor/scanner.py index e69de29..d12ee0e 100644 --- a/serato_doctor/scanner.py +++ b/serato_doctor/scanner.py @@ -0,0 +1,31 @@ +from pathlib import Path + +from serato_doctor.models import DiskTrack + +AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"} + + +def scan_audio(folder: Path) -> list[DiskTrack]: + tracks = [] + + for path in folder.rglob("*"): + if not path.is_file(): + continue + if path.suffix.lower() not in AUDIO_SUFFIXES: + continue + + try: + stat = path.stat() + except OSError: + continue + + tracks.append( + DiskTrack( + path=path, + filename=path.name, + size=stat.st_size, + suffix=path.suffix.lower(), + ) + ) + + return tracks