|
|
|
@@ -1,34 +1,45 @@
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
import re
|
|
|
|
|
|
|
|
|
|
from serato_doctor.models import TrackReference
|
|
|
|
|
|
|
|
|
|
AUDIO_EXTS = "mp3|m4a|wav|aif|aiff|flac|MP3|M4A|WAV|AIF|AIFF|FLAC"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def read_crate_text(crate_path: Path) -> str:
|
|
|
|
|
raw = crate_path.read_bytes()
|
|
|
|
|
for enc in ("utf-16-be", "utf-16-le", "utf-8", "latin1"):
|
|
|
|
|
text = raw.decode(enc, errors="ignore")
|
|
|
|
|
if "Users" in text or "Jukebox" in text:
|
|
|
|
|
return text
|
|
|
|
|
return raw.decode("latin1", errors="ignore")
|
|
|
|
|
return crate_path.read_bytes().decode("utf-16-le", errors="ignore").replace("\x00", "")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def clean_path(raw: str) -> str:
|
|
|
|
|
# Common Serato decode artifacts where final extension char gets merged.
|
|
|
|
|
raw = raw.replace(".mp漳", ".mp3")
|
|
|
|
|
raw = raw.replace(".MP漳", ".MP3")
|
|
|
|
|
raw = raw.replace(".m4愠", ".m4a")
|
|
|
|
|
raw = raw.replace(".M4愠", ".M4A")
|
|
|
|
|
raw = raw.replace(".wa瘠", ".wav")
|
|
|
|
|
raw = raw.replace(".WA瘠", ".WAV")
|
|
|
|
|
raw = raw.replace(".ai映", ".aif")
|
|
|
|
|
raw = raw.replace(".AI映", ".AIF")
|
|
|
|
|
return raw.strip()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def parse_crate(crate_path: Path) -> list[TrackReference]:
|
|
|
|
|
text = read_crate_text(crate_path)
|
|
|
|
|
|
|
|
|
|
# Serato crate files often decode with weird spacing/null-ish characters.
|
|
|
|
|
# This finds paths from /Users/... through the audio extension without
|
|
|
|
|
# greedily scanning the entire file.
|
|
|
|
|
pattern = rf"/?Users/[^\r\n]+?\.(?:{AUDIO_EXTS})"
|
|
|
|
|
|
|
|
|
|
refs = []
|
|
|
|
|
for match in re.finditer(pattern, text):
|
|
|
|
|
raw_path = "/" + match.group(0).lstrip("/")
|
|
|
|
|
raw_path = raw_path.replace("\x00", "")
|
|
|
|
|
path = Path(raw_path)
|
|
|
|
|
|
|
|
|
|
marker = "Users/djsplice/OneDrive/Jukebox/"
|
|
|
|
|
|
|
|
|
|
# Serato record markers seen after paths in UTF-16-LE decoded crate data.
|
|
|
|
|
stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"]
|
|
|
|
|
|
|
|
|
|
for part in text.split(marker)[1:]:
|
|
|
|
|
candidate = marker + part
|
|
|
|
|
|
|
|
|
|
stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1]
|
|
|
|
|
if not stops:
|
|
|
|
|
continue
|
|
|
|
|
|
|
|
|
|
raw_path = "/" + candidate[: min(stops)]
|
|
|
|
|
raw_path = clean_path(raw_path)
|
|
|
|
|
|
|
|
|
|
path = Path(raw_path)
|
|
|
|
|
refs.append(
|
|
|
|
|
TrackReference(
|
|
|
|
|
source=crate_path,
|
|
|
|
|