Compare commits

..

3 Commits

Author SHA1 Message Date
Philip Guzman 577fe1f7a7 Merge improved parser into missing report branch 2026-06-30 09:18:22 -07:00
Philip Guzman 826428f380 Improve Serato crate parser for UTF-16 LE path records 2026-06-30 08:43:11 -07:00
Philip Guzman 4ad5b8a574 Add working crate parser and filesystem scanner 2026-06-29 23:09:52 -07:00
4 changed files with 109 additions and 0 deletions
+3
View File
@@ -11,3 +11,6 @@ reports/
# Never commit personal Serato data
database V2
*.crate
# macOS
.DS_Store
+58
View File
@@ -0,0 +1,58 @@
from pathlib import Path
from serato_doctor.models import TrackReference
def read_crate_text(crate_path: Path) -> str:
return crate_path.read_bytes().decode("utf-16-le", errors="ignore").replace("\x00", "")
def clean_path(raw: str) -> str:
# Common Serato decode artifacts where final extension char gets merged.
raw = raw.replace(".mp漳", ".mp3")
raw = raw.replace(".MP漳", ".MP3")
raw = raw.replace(".m4愠", ".m4a")
raw = raw.replace(".M4愠", ".M4A")
raw = raw.replace(".wa瘠", ".wav")
raw = raw.replace(".WA瘠", ".WAV")
raw = raw.replace(".ai映", ".aif")
raw = raw.replace(".AI映", ".AIF")
return raw.strip()
def parse_crate(crate_path: Path) -> list[TrackReference]:
text = read_crate_text(crate_path)
refs = []
marker = "Users/djsplice/OneDrive/Jukebox/"
# Serato record markers seen after paths in UTF-16-LE decoded crate data.
stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"]
for part in text.split(marker)[1:]:
candidate = marker + part
stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1]
if not stops:
continue
raw_path = "/" + candidate[: min(stops)]
raw_path = clean_path(raw_path)
path = Path(raw_path)
refs.append(
TrackReference(
source=crate_path,
path=path,
filename=path.name,
)
)
return refs
def parse_crates(root: Path) -> list[TrackReference]:
refs = []
for crate in root.rglob("*.crate"):
refs.extend(parse_crate(crate))
return refs
+17
View File
@@ -0,0 +1,17 @@
from dataclasses import dataclass
from pathlib import Path
@dataclass(frozen=True)
class TrackReference:
source: Path
path: Path
filename: str
@dataclass(frozen=True)
class DiskTrack:
path: Path
filename: str
size: int
suffix: str
+31
View File
@@ -0,0 +1,31 @@
from pathlib import Path
from serato_doctor.models import DiskTrack
AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"}
def scan_audio(folder: Path) -> list[DiskTrack]:
tracks = []
for path in folder.rglob("*"):
if not path.is_file():
continue
if path.suffix.lower() not in AUDIO_SUFFIXES:
continue
try:
stat = path.stat()
except OSError:
continue
tracks.append(
DiskTrack(
path=path,
filename=path.name,
size=stat.st_size,
suffix=path.suffix.lower(),
)
)
return tracks