Compare commits

...

3 Commits

Author SHA1 Message Date
Philip Guzman 577fe1f7a7 Merge improved parser into missing report branch 2026-06-30 09:18:22 -07:00
Philip Guzman c3bf111107 Add grouped missing reference report 2026-06-30 09:16:27 -07:00
Philip Guzman 826428f380 Improve Serato crate parser for UTF-16 LE path records 2026-06-30 08:43:11 -07:00
4 changed files with 85 additions and 26 deletions
+3
View File
@@ -11,3 +11,6 @@ reports/
# Never commit personal Serato data # Never commit personal Serato data
database V2 database V2
*.crate *.crate
# macOS
.DS_Store
+10 -7
View File
@@ -1,9 +1,9 @@
from pathlib import Path from pathlib import Path
import argparse import argparse
import csv
from serato_doctor.crate_parser import parse_crates from serato_doctor.crate_parser import parse_crates
from serato_doctor.scanner import scan_audio from serato_doctor.scanner import scan_audio
from serato_doctor.report import write_csv, write_missing_report
def main(): def main():
@@ -11,11 +11,13 @@ def main():
parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_")) parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_"))
parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox")) parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox"))
parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv")) parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv"))
parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt"))
args = parser.parse_args() args = parser.parse_args()
serato = Path(args.serato) serato = Path(args.serato)
music = Path(args.music) music = Path(args.music)
out = Path(args.out) out = Path(args.out)
report = Path(args.report)
refs = parse_crates(serato / "Subcrates") refs = parse_crates(serato / "Subcrates")
disk = scan_audio(music) disk = scan_audio(music)
@@ -31,15 +33,16 @@ def main():
"exists_by_filename": ref.filename in disk_names, "exists_by_filename": ref.filename in disk_names,
}) })
with out.open("w", newline="", encoding="utf-8") as f: missing_count = sum(1 for r in rows if not r["exists_by_filename"])
writer = csv.DictWriter(f, fieldnames=["crate", "serato_path", "filename", "exists_by_filename"])
writer.writeheader() write_csv(rows, out)
writer.writerows(rows) write_missing_report(rows, report)
print(f"Crate references: {len(refs)}") print(f"Crate references: {len(refs)}")
print(f"Disk tracks: {len(disk)}") print(f"Disk tracks: {len(disk)}")
print(f"Missing by filename: {sum(1 for r in rows if not r['exists_by_filename'])}") print(f"Missing by filename: {missing_count}")
print(f"Wrote: {out}") print(f"CSV: {out}")
print(f"Report: {report}")
if __name__ == "__main__": if __name__ == "__main__":
+30 -19
View File
@@ -1,34 +1,45 @@
from pathlib import Path from pathlib import Path
import re
from serato_doctor.models import TrackReference from serato_doctor.models import TrackReference
AUDIO_EXTS = "mp3|m4a|wav|aif|aiff|flac|MP3|M4A|WAV|AIF|AIFF|FLAC"
def read_crate_text(crate_path: Path) -> str: def read_crate_text(crate_path: Path) -> str:
raw = crate_path.read_bytes() return crate_path.read_bytes().decode("utf-16-le", errors="ignore").replace("\x00", "")
for enc in ("utf-16-be", "utf-16-le", "utf-8", "latin1"):
text = raw.decode(enc, errors="ignore")
if "Users" in text or "Jukebox" in text: def clean_path(raw: str) -> str:
return text # Common Serato decode artifacts where final extension char gets merged.
return raw.decode("latin1", errors="ignore") raw = raw.replace(".mp漳", ".mp3")
raw = raw.replace(".MP漳", ".MP3")
raw = raw.replace(".m4愠", ".m4a")
raw = raw.replace(".M4愠", ".M4A")
raw = raw.replace(".wa瘠", ".wav")
raw = raw.replace(".WA瘠", ".WAV")
raw = raw.replace(".ai映", ".aif")
raw = raw.replace(".AI映", ".AIF")
return raw.strip()
def parse_crate(crate_path: Path) -> list[TrackReference]: def parse_crate(crate_path: Path) -> list[TrackReference]:
text = read_crate_text(crate_path) text = read_crate_text(crate_path)
# Serato crate files often decode with weird spacing/null-ish characters.
# This finds paths from /Users/... through the audio extension without
# greedily scanning the entire file.
pattern = rf"/?Users/[^\r\n]+?\.(?:{AUDIO_EXTS})"
refs = [] refs = []
for match in re.finditer(pattern, text):
raw_path = "/" + match.group(0).lstrip("/")
raw_path = raw_path.replace("\x00", "")
path = Path(raw_path)
marker = "Users/djsplice/OneDrive/Jukebox/"
# Serato record markers seen after paths in UTF-16-LE decoded crate data.
stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"]
for part in text.split(marker)[1:]:
candidate = marker + part
stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1]
if not stops:
continue
raw_path = "/" + candidate[: min(stops)]
raw_path = clean_path(raw_path)
path = Path(raw_path)
refs.append( refs.append(
TrackReference( TrackReference(
source=crate_path, source=crate_path,
+42
View File
@@ -0,0 +1,42 @@
from collections import Counter, defaultdict
from pathlib import Path
import csv
def write_missing_report(rows: list[dict], out: Path) -> None:
missing = [r for r in rows if not r["exists_by_filename"]]
crate_counts = Counter(r["crate"] for r in missing)
filename_counts = Counter(r["filename"] for r in missing)
with out.open("w", encoding="utf-8") as f:
f.write("# Serato Doctor Missing Report\n\n")
f.write(f"Total missing references: {len(missing)}\n\n")
f.write("## Missing by crate\n\n")
for crate, count in crate_counts.most_common():
f.write(f"{count:5} {crate}\n")
f.write("\n## Most common missing filenames\n\n")
for filename, count in filename_counts.most_common(100):
f.write(f"{count:5} {filename}\n")
f.write("\n## Detail\n\n")
by_crate = defaultdict(list)
for r in missing:
by_crate[r["crate"]].append(r["filename"])
for crate, names in sorted(by_crate.items()):
f.write(f"\n### {crate}\n")
for name in sorted(set(names)):
f.write(f"- {name}\n")
def write_csv(rows: list[dict], out: Path) -> None:
with out.open("w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter(
f,
fieldnames=["crate", "serato_path", "filename", "exists_by_filename"],
)
writer.writeheader()
writer.writerows(rows)