Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4ad5b8a574 |
+7
-10
@@ -1,9 +1,9 @@
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import argparse
|
import argparse
|
||||||
|
import csv
|
||||||
|
|
||||||
from serato_doctor.crate_parser import parse_crates
|
from serato_doctor.crate_parser import parse_crates
|
||||||
from serato_doctor.scanner import scan_audio
|
from serato_doctor.scanner import scan_audio
|
||||||
from serato_doctor.report import write_csv, write_missing_report
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
@@ -11,13 +11,11 @@ def main():
|
|||||||
parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_"))
|
parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_"))
|
||||||
parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox"))
|
parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox"))
|
||||||
parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv"))
|
parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv"))
|
||||||
parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt"))
|
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
serato = Path(args.serato)
|
serato = Path(args.serato)
|
||||||
music = Path(args.music)
|
music = Path(args.music)
|
||||||
out = Path(args.out)
|
out = Path(args.out)
|
||||||
report = Path(args.report)
|
|
||||||
|
|
||||||
refs = parse_crates(serato / "Subcrates")
|
refs = parse_crates(serato / "Subcrates")
|
||||||
disk = scan_audio(music)
|
disk = scan_audio(music)
|
||||||
@@ -33,16 +31,15 @@ def main():
|
|||||||
"exists_by_filename": ref.filename in disk_names,
|
"exists_by_filename": ref.filename in disk_names,
|
||||||
})
|
})
|
||||||
|
|
||||||
missing_count = sum(1 for r in rows if not r["exists_by_filename"])
|
with out.open("w", newline="", encoding="utf-8") as f:
|
||||||
|
writer = csv.DictWriter(f, fieldnames=["crate", "serato_path", "filename", "exists_by_filename"])
|
||||||
write_csv(rows, out)
|
writer.writeheader()
|
||||||
write_missing_report(rows, report)
|
writer.writerows(rows)
|
||||||
|
|
||||||
print(f"Crate references: {len(refs)}")
|
print(f"Crate references: {len(refs)}")
|
||||||
print(f"Disk tracks: {len(disk)}")
|
print(f"Disk tracks: {len(disk)}")
|
||||||
print(f"Missing by filename: {missing_count}")
|
print(f"Missing by filename: {sum(1 for r in rows if not r['exists_by_filename'])}")
|
||||||
print(f"CSV: {out}")
|
print(f"Wrote: {out}")
|
||||||
print(f"Report: {report}")
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -0,0 +1,47 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
import re
|
||||||
|
|
||||||
|
from serato_doctor.models import TrackReference
|
||||||
|
|
||||||
|
AUDIO_EXTS = "mp3|m4a|wav|aif|aiff|flac|MP3|M4A|WAV|AIF|AIFF|FLAC"
|
||||||
|
|
||||||
|
|
||||||
|
def read_crate_text(crate_path: Path) -> str:
|
||||||
|
raw = crate_path.read_bytes()
|
||||||
|
for enc in ("utf-16-be", "utf-16-le", "utf-8", "latin1"):
|
||||||
|
text = raw.decode(enc, errors="ignore")
|
||||||
|
if "Users" in text or "Jukebox" in text:
|
||||||
|
return text
|
||||||
|
return raw.decode("latin1", errors="ignore")
|
||||||
|
|
||||||
|
|
||||||
|
def parse_crate(crate_path: Path) -> list[TrackReference]:
|
||||||
|
text = read_crate_text(crate_path)
|
||||||
|
|
||||||
|
# Serato crate files often decode with weird spacing/null-ish characters.
|
||||||
|
# This finds paths from /Users/... through the audio extension without
|
||||||
|
# greedily scanning the entire file.
|
||||||
|
pattern = rf"/?Users/[^\r\n]+?\.(?:{AUDIO_EXTS})"
|
||||||
|
|
||||||
|
refs = []
|
||||||
|
for match in re.finditer(pattern, text):
|
||||||
|
raw_path = "/" + match.group(0).lstrip("/")
|
||||||
|
raw_path = raw_path.replace("\x00", "")
|
||||||
|
path = Path(raw_path)
|
||||||
|
|
||||||
|
refs.append(
|
||||||
|
TrackReference(
|
||||||
|
source=crate_path,
|
||||||
|
path=path,
|
||||||
|
filename=path.name,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return refs
|
||||||
|
|
||||||
|
|
||||||
|
def parse_crates(root: Path) -> list[TrackReference]:
|
||||||
|
refs = []
|
||||||
|
for crate in root.rglob("*.crate"):
|
||||||
|
refs.extend(parse_crate(crate))
|
||||||
|
return refs
|
||||||
|
|||||||
@@ -0,0 +1,17 @@
|
|||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class TrackReference:
|
||||||
|
source: Path
|
||||||
|
path: Path
|
||||||
|
filename: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class DiskTrack:
|
||||||
|
path: Path
|
||||||
|
filename: str
|
||||||
|
size: int
|
||||||
|
suffix: str
|
||||||
@@ -1,42 +0,0 @@
|
|||||||
from collections import Counter, defaultdict
|
|
||||||
from pathlib import Path
|
|
||||||
import csv
|
|
||||||
|
|
||||||
|
|
||||||
def write_missing_report(rows: list[dict], out: Path) -> None:
|
|
||||||
missing = [r for r in rows if not r["exists_by_filename"]]
|
|
||||||
|
|
||||||
crate_counts = Counter(r["crate"] for r in missing)
|
|
||||||
filename_counts = Counter(r["filename"] for r in missing)
|
|
||||||
|
|
||||||
with out.open("w", encoding="utf-8") as f:
|
|
||||||
f.write("# Serato Doctor Missing Report\n\n")
|
|
||||||
f.write(f"Total missing references: {len(missing)}\n\n")
|
|
||||||
|
|
||||||
f.write("## Missing by crate\n\n")
|
|
||||||
for crate, count in crate_counts.most_common():
|
|
||||||
f.write(f"{count:5} {crate}\n")
|
|
||||||
|
|
||||||
f.write("\n## Most common missing filenames\n\n")
|
|
||||||
for filename, count in filename_counts.most_common(100):
|
|
||||||
f.write(f"{count:5} {filename}\n")
|
|
||||||
|
|
||||||
f.write("\n## Detail\n\n")
|
|
||||||
by_crate = defaultdict(list)
|
|
||||||
for r in missing:
|
|
||||||
by_crate[r["crate"]].append(r["filename"])
|
|
||||||
|
|
||||||
for crate, names in sorted(by_crate.items()):
|
|
||||||
f.write(f"\n### {crate}\n")
|
|
||||||
for name in sorted(set(names)):
|
|
||||||
f.write(f"- {name}\n")
|
|
||||||
|
|
||||||
|
|
||||||
def write_csv(rows: list[dict], out: Path) -> None:
|
|
||||||
with out.open("w", newline="", encoding="utf-8") as f:
|
|
||||||
writer = csv.DictWriter(
|
|
||||||
f,
|
|
||||||
fieldnames=["crate", "serato_path", "filename", "exists_by_filename"],
|
|
||||||
)
|
|
||||||
writer.writeheader()
|
|
||||||
writer.writerows(rows)
|
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from serato_doctor.models import DiskTrack
|
||||||
|
|
||||||
|
AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"}
|
||||||
|
|
||||||
|
|
||||||
|
def scan_audio(folder: Path) -> list[DiskTrack]:
|
||||||
|
tracks = []
|
||||||
|
|
||||||
|
for path in folder.rglob("*"):
|
||||||
|
if not path.is_file():
|
||||||
|
continue
|
||||||
|
if path.suffix.lower() not in AUDIO_SUFFIXES:
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
stat = path.stat()
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
|
||||||
|
tracks.append(
|
||||||
|
DiskTrack(
|
||||||
|
path=path,
|
||||||
|
filename=path.name,
|
||||||
|
size=stat.st_size,
|
||||||
|
suffix=path.suffix.lower(),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return tracks
|
||||||
|
|||||||
Reference in New Issue
Block a user