Compare commits

..

39 Commits

Author SHA1 Message Date
Philip Guzman 4d04939846 Add duplicate audio hash detection 2026-07-01 16:36:21 -07:00
Philip Guzman ec646479d2 Show detailed batch repair preview 2026-07-01 16:26:04 -07:00
Philip Guzman f3cfed316d Add batch duplicate review workflow 2026-07-01 16:18:30 -07:00
Philip Guzman 4315d8d4d4 Add duplicate audio comparison controls 2026-07-01 16:12:35 -07:00
Philip Guzman 139822f8a5 Fix recovery layout and interactive assets 2026-07-01 16:03:28 -07:00
Philip Guzman da72419ce9 Add backup recovery center 2026-07-01 15:34:58 -07:00
Philip Guzman c4f1b4535d Add backup-first duplicate repair 2026-07-01 15:29:44 -07:00
Philip Guzman d2f625ed18 Add diagnostic drilldowns 2026-07-01 09:17:25 -07:00
Philip Guzman fb3d70e579 Clarify diagnostic language and explanations 2026-07-01 08:37:18 -07:00
Philip Guzman ce37b45058 Merge feature/database-v2-parser into develop 2026-07-01 08:30:53 -07:00
Philip Guzman 37786612b7 Add read-only database V2 parser 2026-07-01 08:13:20 -07:00
Philip Guzman 8ff4db4951 Merge feature/smart-crate-discovery into develop 2026-07-01 08:08:43 -07:00
Philip Guzman a384cdc88d Discover Serato smart crate definitions 2026-07-01 08:06:19 -07:00
Philip Guzman 71bca064ed Merge feature/web-interface into develop 2026-07-01 08:03:19 -07:00
Philip Guzman eda5d4e62b Document approved dashboard direction 2026-07-01 07:55:00 -07:00
Philip Guzman 9053b4ca3d Add local web analysis dashboard 2026-07-01 07:52:33 -07:00
Philip Guzman 251ab4d090 Merge feature/broken-symlinks into develop 2026-07-01 07:46:30 -07:00
Philip Guzman b8ea17450b Add broken symlink diagnostics 2026-07-01 07:32:07 -07:00
Philip Guzman c5449a278a Merge feature/duplicate-filenames into develop 2026-07-01 07:30:50 -07:00
Philip Guzman 6449c7de6a Add duplicate filename diagnostics 2026-06-30 18:52:16 -07:00
Philip Guzman d764f910e4 Merge feature/crate-classification into develop 2026-06-30 18:50:42 -07:00
Philip Guzman 35713bbbb3 Classify static and smart crates 2026-06-30 18:20:05 -07:00
Philip Guzman c92d929f37 Merge feature/analyze-command into develop 2026-06-30 18:17:20 -07:00
Philip Guzman e6f313d98a Add read-only analyze command 2026-06-30 17:55:45 -07:00
Philip Guzman 22424d0b5c Merge feature/health-engine into develop 2026-06-30 17:54:20 -07:00
Philip Guzman b8bf9ec6e1 Add transparent library health engine 2026-06-30 17:47:54 -07:00
Philip Guzman 167f029c28 Merge feature/matching-engine into develop 2026-06-30 17:46:56 -07:00
Philip Guzman 5a2edb8b94 Add explainable matching engine 2026-06-30 17:42:33 -07:00
Philip Guzman 29e53f7dc6 Merge feature/logging into develop 2026-06-30 17:40:57 -07:00
Philip Guzman 4fde93613f Add opt-in diagnostic logging 2026-06-30 17:36:58 -07:00
Philip Guzman e4d5a32de2 Merge feature/configuration into develop 2026-06-30 17:35:59 -07:00
Philip Guzman ec671079ba Add configurable library reference roots 2026-06-30 17:20:08 -07:00
Philip Guzman 4e85b5912b Merge feature/sample-library into develop 2026-06-30 17:18:06 -07:00
Philip Guzman d4fe61cf8c Merge feature/test-foundation into develop 2026-06-30 17:18:06 -07:00
Philip Guzman c4d08bfe27 Merge feature/library-model into develop 2026-06-30 17:18:06 -07:00
Philip Guzman bb74c14742 Add synthetic sample library 2026-06-30 17:15:56 -07:00
Philip Guzman 71fa64a960 Establish automated test foundation 2026-06-30 10:31:34 -07:00
Philip Guzman 09167e44c7 Introduce core library model 2026-06-30 10:10:14 -07:00
Philip Guzman 508eecd0bd Add project roadmap and architecture docs 2026-06-30 09:53:09 -07:00
74 changed files with 4701 additions and 65 deletions
+3
View File
@@ -2,11 +2,14 @@ __pycache__/
*.pyc *.pyc
.env .env
.venv/ .venv/
*.egg-info/
reports/ reports/
samples/small-library/generated/
*.sqlite *.sqlite
*.db *.db
*.csv *.csv
*.html *.html
!serato_doctor/webui/*.html
# Never commit personal Serato data # Never commit personal Serato data
database V2 database V2
+19
View File
@@ -0,0 +1,19 @@
# Contributing
## Branching
- `main` is stable.
- `develop` is the integration branch.
- Feature branches use: `feature/<name>`.
## Safety Rules
Never commit personal music library files, Serato databases, or crates.
Never write repair code without dry-run mode, backup plan, and rollback log.
## Testing
Install development dependencies with `python3 -m pip install -e '.[dev]'`.
Run `pytest` before committing. Tests and fixtures must use synthetic library data.
+48
View File
@@ -0,0 +1,48 @@
# Serato Doctor
Serato Doctor is a read-only-first toolkit for inspecting and eventually repairing
DJ libraries. Serato is the first supported engine.
## Analyze a Library
```shell
serato-doctor analyze \
--serato ~/Music/_Serato_ \
--music ~/Music/Jukebox
```
Analysis prints a transparent reference-integrity score plus broken references,
duplicate filenames, unused tracks, and suggested filename matches. It does not
modify the library or create report files.
If crates contain an older library root, supply it explicitly:
```shell
serato-doctor analyze --reference-root /Users/old-user/OneDrive/Jukebox
```
## Generate Detailed Reports
Running without a command preserves the original scanner behavior and writes CSV
and text reports:
```shell
serato-doctor --serato ~/Music/_Serato_ --music ~/Music/Jukebox
```
Use `--verbose` for progress on standard error or `--log-file PATH` for an
aggregate diagnostic log.
Serato Doctor never repairs files without an explicit future repair workflow,
preview, backup, and rollback path.
## Local Web Interface
Launch the responsive, local-only dashboard with:
```shell
serato-doctor-web
```
Then open `http://127.0.0.1:8765`. The interface exposes the same read-only health
analysis and never sends library paths or results to an external service.
+52
View File
@@ -0,0 +1,52 @@
# Serato Doctor Roadmap
## v0.1 — Library Inspector
- [x] Project repository
- [x] Filesystem scanner
- [x] Serato crate parser
- [x] Missing reference CSV report
- [x] Grouped missing reference report
- [x] HTML health dashboard
- [x] Test suite
- [x] Sample library fixtures
- [x] Database V2 read-only parser
- [x] Configuration
- [x] Logging
- [x] Matching engine
## v0.2 — Diagnostics
- [x] `serato-doctor analyze` command
- [x] Duplicate filename detection
- [x] Duplicate audio hash detection
- [x] Broken symlink detection
- [ ] Orphaned audio detection
- [ ] OneDrive rename detection
- [x] Crate classification: static vs smart/dynamic
- [x] Library health score
## v0.3 — Safe Repair
- [x] Dry-run repair plan
- [x] Backup before repair
- [x] Compatibility symlink creation
- [ ] Compatibility copy creation
- [ ] Rename repair
- [x] Rollback log and recovery center
## v0.4 — Migration Wizard
- [ ] Move library root
- [ ] Cloud provider migration
- [ ] External drive migration
- [ ] Verify moved library
- [ ] Update application references
## v1.0 — DJ Library Doctor
- [ ] Desktop UI
- [ ] Serato support
- [ ] Rekordbox support
- [ ] VirtualDJ support
- [ ] Engine DJ support
+11
View File
@@ -0,0 +1,11 @@
# Architecture
Serato Doctor is designed as a DJ library inspection, repair, and migration platform.
## Design Principles
1. Read-only by default.
2. Every repair must support preview/dry-run.
3. Every repair must create a backup or rollback path.
4. Application-specific logic lives in engines.
5. Core matching and scanning logic should be application-agnostic.
@@ -0,0 +1,26 @@
# Case Study: OneDrive Mac Migration
## Scenario
A large Serato DJ library was migrated from an older Mac to a newer Mac using OneDrive.
## Symptoms
- OneDrive client stuck syncing
- Duplicate OneDrive folders
- Thousands of files renamed with trailing ` 2`
- Serato reported many tracks as missing
- Some files existed on disk but still appeared orange in Serato
## Findings
- OneDrive sync state was rebuilt successfully
- Thousands of orphaned filename conflicts were repaired
- Some Serato references were stale database objects, not missing files
- Smart/dynamic crates should be classified separately from static user crates
## Lessons
- Filesystem health and Serato database health are separate problems
- Smart crates should not be treated the same as static crates
- Repair tools must be read-only by default and generate a plan before changing anything
+27
View File
@@ -0,0 +1,27 @@
# Analyze Command
## Problem
The health and matching engines are only Python APIs. Users need one safe command
that summarizes library integrity without first interpreting CSV files.
## Architecture
`serato-doctor analyze` reuses the existing configured crate parser and filesystem
scanner, then prints the immutable health report. The original no-command mode is
retained as the report-producing scan workflow.
Analyze mode does not create CSV or text reports. Both modes remain read-only with
respect to Serato crates, databases, and music files.
## Edge Cases
- A library with no references displays `Not assessed` rather than a false score.
- Historical roots continue to work through repeatable `--reference-root` flags.
- Normal output remains separate from optional diagnostic logging.
- Existing scripts that invoke the CLI without a subcommand remain compatible.
## Verification
Tests prove the legacy five-line output and reports remain unchanged, while analyze
prints health metrics and does not create output files.
+32
View File
@@ -0,0 +1,32 @@
# Audio comparison for duplicate review
## Problem
Filenames alone are weak evidence. DJs need to hear each candidate and locate it
on disk before deciding which copy is authoritative.
## Design
Exact-duplicate and cloud-conflict groups expose a shared audio player with a
preview action for each file. Switching files reuses the same player, making
back-to-back comparison quick. A separate action reveals the selected file in
macOS Finder.
The browser never receives unrestricted filesystem access. Each analyzed audio
path gets a short, process-local HMAC token. Preview and Finder endpoints reject
altered, expired, missing, non-audio, or otherwise unsigned paths. Audio serving
supports HTTP byte ranges so playback can seek without loading an entire track.
## Edge cases
- A file moved after analysis is rejected.
- Forged or stale tokens cannot select another local file.
- Preview controls remain useful if autoplay is blocked because native audio
controls stay visible.
- Finder failures are reported beside the button without affecting analysis.
## Tests
- Valid signed audio resolves to the analyzed file.
- A modified token is rejected.
- Duplicate detail payloads include preview and Finder controls for every file.
+30
View File
@@ -0,0 +1,30 @@
# Backup recovery center
## Problem
An immediate undo button is not enough. A DJ may restart Serato Doctor, notice
an issue later, or need to understand how much disk space safety snapshots use.
## Architecture
The recovery center reads the existing JSON manifests beneath
`_Serato_/.serato-doctor-backups/`. It reports creation time, selected keeper,
affected paths, restore status, and snapshot size. Invalid or incomplete backup
folders are ignored rather than presented as restorable.
Restore remains conservative: it only replaces a symbolic-link alias and
refuses to overwrite a real file. A successful restore records its timestamp in
the manifest so the UI cannot accidentally offer the same rollback twice.
## Edge cases
- No backup directory is treated as an empty history.
- Malformed manifests do not prevent valid backups from loading.
- Restored snapshots remain visible for audit purposes and disk accounting.
- Backups can only be restored through the Serato library that owns them.
## Tests
- Backup history reports size and ready status.
- Restore changes the persistent status to restored.
- Static recovery assets are included in the package.
+37
View File
@@ -0,0 +1,37 @@
# Batch duplicate review
## Problem
Opening a diagnostic, opening each group again, scrolling to a separate repair
form, and applying one change at a time creates unnecessary friction and many
small backups.
## Design
Duplicate and cloud-conflict diagnostics now open as a sequential review queue.
Each group presents its candidates together with audio preview and Finder
controls. Choosing a keeper records the decision and advances immediately.
Uncertain groups can be skipped without changing them.
All approved decisions are submitted as one dry-run plan. Apply creates one
backup containing every affected audio file plus the Serato metadata snapshot,
then installs all compatibility shortcuts as one rollback unit. Any validation
or filesystem failure prevents or rolls back the batch.
The dry-run opens in a closable modal that names every selected keeper and every
path that will become a compatibility shortcut. Closing the modal preserves the
queue position and all decisions so the DJ can continue reviewing before apply.
## Edge cases
- The same group cannot be submitted twice.
- A file cannot be both a keeper and a disposable file across choices.
- Every group is rescanned and revalidated before preview and apply.
- Skipped and unreviewed groups remain untouched.
- Revisiting a group preserves and visibly marks its selected keeper.
## Tests
- Two groups produce one backup and one restorable receipt.
- Batch preview reports combined changes without modifying files.
- Existing single-group repair remains a wrapper around the batch engine.
+27
View File
@@ -0,0 +1,27 @@
# Broken Symlink Detection
## Problem
Library migrations may leave symbolic links pointing to files or folders that no
longer exist. The filesystem scanner previously skipped those links silently.
## Architecture
One read-only filesystem traversal now returns audio tracks and broken symbolic
links. Each finding preserves the link path and its raw target when the operating
system can read it. The library and health report retain only immutable findings.
Broken links are reported separately and do not affect the reference-integrity
score. The detector does not follow, recreate, remove, or rewrite any link.
## Edge Cases
- Relative and absolute link targets.
- Links to missing files and missing directories.
- Link targets that cannot be read due to an operating-system error.
- Valid symlinks, which remain eligible for ordinary audio scanning.
## Verification
Tests create synthetic valid and broken links in temporary directories, verify the
raw target, and confirm health integration. Existing scanner behavior is preserved.
+25
View File
@@ -0,0 +1,25 @@
# Scan Configuration
## Problem
The prototype recognized crate paths by matching one developer's historical
OneDrive path. That made otherwise valid libraries invisible to the parser.
## Architecture
`ScanConfig` owns the paths for a read-only scan. The CLI accepts repeatable
`--reference-root` options for roots embedded in crates before a migration. The
parser uses those roots as record boundaries. Without explicit roots, it recognizes
generic macOS home and mounted-volume paths without embedding a username.
## Edge Cases
- A library may have references from more than one historical root.
- Root paths may contain spaces or have leading/trailing separators.
- Existing `/Users/...` and `/Volumes/...` crates must work without new flags.
- A record without a recognized terminator remains ignored.
## Verification
Tests cover default root discovery, a custom migrated root, the CLI option, and
the existing synthetic sample library. The scanner remains read-only.
+38
View File
@@ -0,0 +1,38 @@
# Crate Classification
## Problem
Static crates are manually maintained track lists. Smart crates are dynamic views
generated from rules, so stale-looking entries in them should not be presented as
broken manual references or given the same health-score weight.
## Architecture
Crates carry a `static`, `smart`, or `unknown` kind. Folder provenance is the
primary signal: Serato stores regular `.crate` files in `Subcrates` and smart
`.scrate` definitions in `SmartCrates`. `Compatible by key.crate` is also treated as smart
when encountered in `Subcrates`, based on the original migration case study.
Smart crate names use `≫≫` to encode hierarchy. The model preserves those segments,
so `Compatible by key≫≫10A.scrate` has a parent of `Compatible by key` and a display
name of `10A`. Smart definitions and dynamic `.crate` containers are counted
separately.
The library loader reads both folders. Health analysis reports all references but
scores only non-smart references. Unknown crates remain scoreable so incomplete
classification cannot silently hide potential problems.
Serato documents the folder distinction in [What is in the _Serato_ folder?](https://support.serato.com/hc/en-us/articles/204022904-What-is-in-the-Serato-folder)
and explains that smart crates are populated from rules in [Crates in Serato DJ](https://support.serato.com/hc/en-us/articles/227561407-Crates-in-Serato-DJ-Pro-Serato-DJ-Lite).
## Edge Cases
- The known `Compatible by key` dynamic crate in the `Subcrates` folder.
- Crate fixtures outside a recognized Serato folder.
- Libraries containing both static and smart references to the same track.
- Dynamic references whose current materialized paths appear missing.
## Verification
Tests cover all three kinds, both Serato folders, the known dynamic fallback, the
synthetic five-static/two-smart library, and exclusion from health scoring.
+28
View File
@@ -0,0 +1,28 @@
# Database V2 Read-only Parser
## Problem
Crates and files do not explain every orange track in Serato. The legacy
`database V2` contains Serato's library-level track paths and metadata, so it must
be inspected independently from crate references.
## Architecture
The parser reads the file as a big-endian tag-length-value stream. Top-level
`otrk` records contain nested fields including `pfil` (path), `tsng` (title),
`tart` (artist), `talb` (album), and `tgen` (genre). Text is UTF-16 big-endian.
Analysis reports total database entries, entries whose filenames occur in the
selected music scan, entries outside that scan, scanned tracks absent from the
database, and duplicate database paths. “Outside scan” is deliberately not called
missing because Serato databases can include samples and tracks from other roots.
## Safety
The parser calls only `read_bytes`; it never opens the database for writing. No
metadata values or personal paths are sent to logs or the dashboard.
## Verification
Synthetic TLV fixtures cover version, paths, metadata, incomplete records, and
health integration. The sample library includes a generated ten-entry database.
+35
View File
@@ -0,0 +1,35 @@
# Diagnostic Drill-Downs
## Problem
Aggregate health numbers are useful, but DJs need to understand what is behind
each number before they can trust it. A count like "60 missing tracks" should be
clickable enough to answer: which tracks, which saved paths, and why did Serato
Doctor count them?
## Architecture
The health engine remains responsible for aggregate scoring. The web layer adds
a separate `details` payload beside the existing health report so the UI can show
examples without changing the core score model.
The drill-down payload is intentionally capped. Serato Doctor should explain the
finding quickly in the local browser, not dump an entire user library into the
page.
## Edge Cases
- Smart/dynamic crate references stay excluded from old-reference scoring.
- The same missing filename can appear more than once in Serato's database.
- Duplicate filename and cloud-conflict groups are informational only.
- Suggested matches remain read-only evidence and must never trigger repair.
- Empty drill-downs should feel reassuring, not broken.
## Testing
- Web analysis should include detail sections for old crate references,
suggested matches, unused tracks, duplicates, cloud conflicts, broken
symlinks, and missing Serato database tracks.
- Static assets should expose clickable diagnostic hooks.
- Browser behavior should be verified manually when the local browser policy
allows access to the development server.
+34
View File
@@ -0,0 +1,34 @@
# Duplicate audio hashing
## Problem
Matching filenames do not prove matching content. Two files can be different DJ
edits, masters, encodes, or entirely different tracks. Removing either without
stronger evidence is unsafe.
## Design
Serato Doctor computes SHA-256 only for the duplicate group currently being
reviewed. This avoids hashing an entire 22,000-track library during every scan.
The review queue labels a group as byte-for-byte identical or warns that its
files differ. File size and a short fingerprint remain available as supporting
detail.
The final batch dry-run recomputes every approved comparison server-side and
includes the status beside each keeper decision. The hash is evidence, not an
automatic repair decision; the DJ remains in control.
## Edge cases
- Identical audio stored under different filenames is recognized.
- Metadata changes inside an audio container produce a different byte hash and
therefore a conservative warning.
- Missing or unreadable files fail verification rather than being treated as
identical.
- Hashing is streamed in chunks rather than loading whole tracks into memory.
## Tests
- Equal byte content produces identical SHA-256 fingerprints.
- Different content is labeled different.
- Batch previews carry the server-computed confidence status.
+32
View File
@@ -0,0 +1,32 @@
# Duplicate Filename Detection
## Problem
Two files with the same filename may be ordinary copies, while names such as
`Track.mp3` and `Track 2.mp3` may indicate a OneDrive conflict. These cases need
review, but neither is sufficient evidence for deletion.
## Architecture
The duplicate detector emits immutable groups of two kinds:
- `exact_name` groups filenames after case and Unicode normalization.
- `cloud_conflict` groups names after additionally removing a trailing numeric
suffix from the stem.
Groups contain every path, a stable comparison key, a display name, and the number
of files beyond the first. Health analysis reports exact and suspected-conflict
counts separately. Neither category changes the health score.
## Edge Cases
- Same filename in different folders.
- Case-only and Unicode representation differences.
- Multiple conflict suffixes such as `Track 2.mp3` and `Track 3.mp3`.
- Legitimate numbered titles, which remain explicitly labeled as suspected.
- A single file, which is never reported as a duplicate.
## Verification
Tests cover exact, normalized, conflict-family, unrelated, and deterministic-order
behavior, plus integration with health analysis. Detection is read-only.
+39
View File
@@ -0,0 +1,39 @@
# Duplicate repair
## Problem
Duplicate and cloud-conflict files waste space, but deleting either path can
unmap tracks in crates or Serato's database. DJs need to choose the authoritative
copy and understand every change before it happens.
## Design
The web interface requires an analysis, an explicit keeper selection, and a
dry-run preview. Applying the plan first creates a timestamped backup beneath
`_Serato_/.serato-doctor-backups/`. The snapshot contains every replaced audio
file, loaded crate/smart-crate metadata, database V2, and a JSON restore manifest.
The non-kept audio path is then replaced with a symbolic link to the keeper.
This removes the extra audio payload while preserving every existing saved path.
Crate files and database V2 are never rewritten. The UI exposes immediate
restore using the manifest.
Users may retain all backups or set a positive rotation limit. Rotation occurs
only after a repair succeeds.
## Edge cases
- The duplicate group is rescanned and validated immediately before preview and
apply.
- Existing symlinks cannot be selected as disposable duplicate files.
- A partial failure restores already-changed files before reporting the error.
- Restore refuses to overwrite a real file.
- Healthy symlink aliases are excluded from future duplicate counts.
## Tests
- Backup creation includes audio and Serato metadata.
- The old path resolves to the selected keeper after repair.
- Restore returns the original file contents.
- Invalid keeper choices are rejected.
- Limited and unlimited retention behave deterministically.
+32
View File
@@ -0,0 +1,32 @@
# Library Health Engine
## Problem
Raw missing-reference counts do not provide a compact view of library integrity,
but an opaque blended score would imply confidence the current data cannot support.
## Architecture
The health engine produces an immutable report from the core `Library`. Its score
is only the percentage of non-dynamic crate references resolved by exact filename.
Smart-crate references are counted but excluded because their contents are derived
from rules. The report also exposes missing references, unique missing filenames,
duplicate filename groups, extra duplicate files, unused tracks, and missing
references with matching candidates.
Duplicate, unused, and candidate counts are informational. They do not affect the
score until the project has a documented and validated weighting policy. An empty
library has no score rather than a misleading 0% or 100%.
## Edge Cases
- The same missing filename referenced by several crates.
- Several disk files sharing a filename.
- Conflict-suffixed files that are unused but may be match candidates.
- Libraries with no crate references.
- Tracks referenced by filename from more than one crate.
## Verification
Tests assert every metric, the disclosed score basis, candidate integration, and
empty-library behavior. Analysis remains entirely read-only.
+25
View File
@@ -0,0 +1,25 @@
# Application Logging
## Problem
The CLI reports final counts but provides no diagnostic trail when a scan behaves
unexpectedly. Troubleshooting should not require adding print statements or expose
library contents by default.
## Architecture
The project uses an isolated standard-library logger. It has no visible output by
default. `--verbose` writes progress to standard error, while `--log-file PATH`
writes an informational audit trail. Normal result lines remain on standard output.
## Edge Cases
- Reconfiguring logging in the same process must not duplicate handlers.
- Console and file logging may be enabled together.
- Log messages contain aggregate counts, not track names or crate contents.
- A default scan must remain quiet except for its established result output.
## Verification
Tests verify quiet defaults, file output, handler replacement, and unchanged CLI
result lines. The full sample-library scan remains read-only.
+29
View File
@@ -0,0 +1,29 @@
# Explainable Matching Engine
## Problem
Missing references need ranked candidate files, but a filename-only yes/no check
cannot explain ambiguity or cloud-provider conflict names.
## Architecture
The read-only matching engine indexes normalized filenames and scores only related
candidates. Every score contains evidence for filename, extension, and parent
folder. Exact filenames earn 60 points, normalized names 55, numeric conflict-name
matches 50, extensions 10, and parent folders 20.
The displayed percentage is an evidence score, not a statistical probability.
Metadata, duration, hashes, and fingerprints can add stronger evidence later.
## Edge Cases
- Unicode and case differences.
- OneDrive-style names such as `Track 2.mp3`.
- Duplicate candidates in different folders.
- Legitimate numbered song titles, which remain candidates but are never repaired.
- Unrelated names, which are not emitted as candidates.
## Verification
Tests cover exact, normalized, conflict-suffix, ambiguous, and unrelated filenames.
Candidate ordering is deterministic. The engine never changes a track or reference.
+25
View File
@@ -0,0 +1,25 @@
# Plain-language Diagnostics
## Problem
“Broken references” combined crate occurrences with Serato's own missing-track
concept. DJs naturally compared that number with orange/unmapped tracks in Serato,
even though the two counts describe different layers.
## Language
- **Missing tracks in Serato** means database entries whose saved file location no
longer exists. This corresponds most closely to orange or unmapped tracks.
- **Old crate references** means saved appearances in regular crates whose exact
filename was not found in the selected music folder. One track can appear in
several crates, so both appearances and unique filenames are shown.
Every health card and diagnostic row has an accessible information button. Hover
shows its explanation on desktop; click or tap keeps it open; Escape or clicking
elsewhere closes it. Explanations describe uncertainty and avoid implying repair.
## Verification
Health tests verify database missing-path and unique-filename counts. Browser tests
cover the two separate metrics, hover/click explanations, keyboard dismissal, and
mobile layout.
+26
View File
@@ -0,0 +1,26 @@
# Synthetic Sample Library
## Problem
Real Serato libraries contain private paths, listening history, and copyrighted
music. Contributors still need a repeatable library for tests and demonstrations.
## Architecture
`samples/small-library/manifest.json` describes a compact migration scenario.
`generate.py` turns that manifest into fake audio files and UTF-16-LE crate files
inside an ignored `generated` directory. The generated files are disposable and
contain no real audio or user library data.
## Edge Cases
- Five manually maintained crates and two smart/dynamic crates.
- Mixed supported audio extensions.
- One reference whose file is absent.
- One stale reference whose file has been renamed.
- References shared by static and smart crates.
## Verification
`pytest` generates the sample in a temporary directory and verifies its crate,
track, and missing-reference counts through the production parser and scanner.
+19
View File
@@ -0,0 +1,19 @@
# Smart Crate Discovery Correction
## Problem
The first classifier searched `SmartCrates` for `*.crate`. Real Serato smart-crate
definitions use `.scrate`, so a library with 26 definitions displayed only the
single name-based `Compatible by key.crate` fallback.
## Correction
Library discovery now reads `.scrate` definitions case-insensitively from the
`SmartCrates` folder. Dynamic `.crate` containers remain excluded from manual
reference scoring but are reported separately. The `≫≫` filename separator is
preserved as smart-crate hierarchy metadata.
## Verification
Synthetic tests cover `.scrate` discovery, case-correct folder names, hierarchy,
definition/container counts, and the existing five-static/two-smart sample.
+24
View File
@@ -0,0 +1,24 @@
# Test Foundation
## Problem
The crate parser handles unusual binary text and known extension artifacts. Without
automated tests, a small refactor could silently change library counts or reports.
## Architecture
Tests cover the read-only pipeline from crate and filesystem inputs through the
core library model to CSV, text, and CLI output. All test data is synthetic and is
created in temporary directories.
## Edge Cases
- Known MP3, M4A, WAV, and AIF decode artifacts.
- Crate records without a recognized stop marker.
- Supported extensions with mixed case and unsupported files.
- References that exist by filename and references that remain missing.
## Verification
Run `pytest`. No test reads a real Serato library or writes outside pytest's
temporary directory.
Binary file not shown.

After

Width:  |  Height:  |  Size: 1.2 MiB

+37
View File
@@ -0,0 +1,37 @@
# Local Web Interface
![Approved Serato Doctor dashboard concept](web-dashboard-concept.png)
The generated concept above is the approved visual direction. The implemented UI
keeps its hierarchy, palette, safety emphasis, health ring, diagnostic cards, and
responsive behavior. Product truth takes precedence over mockup copy: a 77.8%
library is labeled for review rather than described as healthy.
## Problem
The command line is useful for automation but makes the growing diagnostic set
harder to explore. A visual dashboard lets users test analysis safely and understand
which findings affect health.
## Architecture
`serato-doctor-web` binds to `127.0.0.1:8765` by default and serves package-owned
HTML, CSS, and JavaScript with Python's standard library. A same-origin JSON endpoint
runs the existing read-only parser, scanner, matcher, duplicate detector, crate
classifier, and health engine. No web framework or external service is required.
The UI clearly labels read-only mode, separates scored health from informational
diagnostics, and adapts from a full sidebar layout to compact mobile navigation.
## Edge Cases
- Missing or invalid Serato and music directories.
- Empty libraries with no assessable health score.
- Large or malformed requests, capped at 64 KiB.
- HTML injection, avoided by rendering all results through `textContent`.
- Network exposure, avoided by a loopback-only default binding.
## Verification
Tests exercise the web analysis adapter and static package assets. Browser checks
cover real form submission, result rendering, error display, and responsive layout.
+26
View File
@@ -0,0 +1,26 @@
[build-system]
requires = ["setuptools>=61"]
build-backend = "setuptools.build_meta"
[project]
name = "serato-doctor"
version = "0.1.0"
description = "Inspect, diagnose, repair, and migrate DJ libraries."
requires-python = ">=3.9"
dependencies = []
[project.optional-dependencies]
dev = ["pytest>=8,<9"]
[project.scripts]
serato-doctor = "serato_doctor.cli:main"
serato-doctor-web = "serato_doctor.web:main"
[tool.pytest.ini_options]
testpaths = ["tests"]
[tool.setuptools.packages.find]
include = ["serato_doctor*"]
[tool.setuptools.package-data]
"serato_doctor.webui" = ["*.html", "*.css", "*.js"]
+25
View File
@@ -0,0 +1,25 @@
# Small Synthetic Library
This fixture models a small Serato migration without containing music or personal
library data. It has 10 fake audio files, five static crates, two smart crates,
one absent song, and one renamed song.
Generate it with:
```shell
python3 samples/small-library/generate.py
```
Analyze it with:
```shell
python3 -m serato_doctor.cli \
--serato samples/small-library/generated/Serato/_Serato_ \
--music samples/small-library/generated/Music \
--out samples/small-library/generated/scan.csv \
--report samples/small-library/generated/report.txt
```
Expected CLI counts are 13 crate references, 10 disk tracks, and 2 references
missing by filename. Everything under `generated/` is disposable and ignored by
Git.
+80
View File
@@ -0,0 +1,80 @@
"""Generate a disposable, synthetic Serato library from the sample manifest."""
import argparse
import json
from pathlib import Path
from typing import Optional
SAMPLE_ROOT = Path(__file__).parent
SERATO_PATH_PREFIX = "Users/sample-user/OneDrive/Jukebox/"
def database_record(tag: bytes, payload: bytes) -> bytes:
return tag + len(payload).to_bytes(4, "big") + payload
def load_manifest() -> dict:
return json.loads((SAMPLE_ROOT / "manifest.json").read_text(encoding="utf-8"))
def build_sample(output: Optional[Path] = None) -> Path:
root = output or SAMPLE_ROOT / "generated"
manifest = load_manifest()
music_root = root / "Music"
serato_root = root / "Serato" / "_Serato_"
for relative_path in manifest["tracks"]:
track_path = music_root / relative_path
track_path.parent.mkdir(parents=True, exist_ok=True)
track_path.write_bytes(
f"Synthetic Serato Doctor fixture: {relative_path}\n".encode("utf-8")
)
for crate in manifest["crates"]:
is_smart = crate["type"] == "smart"
folder_name = "SmartCrates" if is_smart else "Subcrates"
crate_root = serato_root / folder_name
crate_root.mkdir(parents=True, exist_ok=True)
output_name = (
Path(crate["name"]).with_suffix(".scrate").name
if is_smart
else crate["name"]
)
for other_folder in ("Subcrates", "Smartcrates", "SmartCrates"):
for stale_name in (
crate["name"],
Path(crate["name"]).with_suffix(".scrate").name,
):
stale_path = serato_root / other_folder / stale_name
if stale_path.exists() and stale_path != crate_root / output_name:
stale_path.unlink()
records = "".join(
f"{SERATO_PATH_PREFIX}{relative_path}otrk"
for relative_path in crate["references"]
)
(crate_root / output_name).write_bytes(records.encode("utf-16-le"))
database_records = [
database_record(b"vrsn", "2.0/Serato Doctor Fixture".encode("utf-16-be"))
]
for relative_path in manifest["tracks"]:
fields = database_record(
b"pfil", str(music_root / relative_path).encode("utf-16-be")
)
database_records.append(database_record(b"otrk", fields))
(serato_root / "database V2").write_bytes(b"".join(database_records))
return root
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--output", type=Path)
args = parser.parse_args()
root = build_sample(args.output)
print(f"Generated synthetic library: {root}")
if __name__ == "__main__":
main()
+56
View File
@@ -0,0 +1,56 @@
{
"tracks": [
"House/First.mp3",
"House/Second.m4a",
"Open Format/Third.wav",
"Open Format/Fourth.aif",
"Classics/Fifth.mp3",
"Classics/Sixth.flac",
"Warmup/Seventh.mp3",
"Warmup/Eighth.mp3",
"Renamed/New Name.mp3",
"Bonus/Ninth.MP3"
],
"crates": [
{
"name": "House.crate",
"type": "static",
"references": ["House/First.mp3", "House/Second.m4a", "House/Missing.mp3"]
},
{
"name": "Open Format.crate",
"type": "static",
"references": ["Open Format/Third.wav", "Open Format/Fourth.aif"]
},
{
"name": "Classics.crate",
"type": "static",
"references": ["Classics/Fifth.mp3", "Classics/Sixth.flac"]
},
{
"name": "Renamed Tracks.crate",
"type": "static",
"references": ["Renamed/Old Name.mp3"]
},
{
"name": "Bonus.crate",
"type": "static",
"references": ["Bonus/Ninth.MP3"]
},
{
"name": "Compatible by key.crate",
"type": "smart",
"references": ["House/First.mp3", "House/Second.m4a"]
},
{
"name": "Smart Warmup.crate",
"type": "smart",
"references": ["Warmup/Seventh.mp3", "Warmup/Eighth.mp3"]
}
],
"expected": {
"disk_tracks": 10,
"crate_references": 13,
"missing_by_filename": ["Missing.mp3", "Old Name.mp3"]
}
}
+110 -31
View File
@@ -1,48 +1,127 @@
from pathlib import Path from pathlib import Path
import argparse import argparse
from serato_doctor.crate_parser import parse_crates from serato_doctor.config import ScanConfig
from serato_doctor.scanner import scan_audio from serato_doctor.crate_parser import load_library_crates
from serato_doctor.database_parser import parse_database
from serato_doctor.health import analyze_health
from serato_doctor.logging import configure_logging
from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_filesystem
from serato_doctor.report import write_csv, write_missing_report from serato_doctor.report import write_csv, write_missing_report
def main(): def main():
parser = argparse.ArgumentParser(prog="serato-doctor") parser = argparse.ArgumentParser(prog="serato-doctor")
parser.add_argument(
"command", nargs="?", choices=("scan", "analyze"), default="scan"
)
parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_")) parser.add_argument("--serato", default=str(Path.home() / "Music/_Serato_"))
parser.add_argument("--music", default=str(Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox")) parser.add_argument(
parser.add_argument("--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv")) "--music",
parser.add_argument("--report", default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt")) default=str(
Path.home() / "Library/CloudStorage/OneDrive-Personal/Jukebox"
),
)
parser.add_argument(
"--out", default=str(Path.home() / "Desktop/serato_doctor_scan.csv")
)
parser.add_argument(
"--report",
default=str(Path.home() / "Desktop/serato_doctor_missing_report.txt"),
)
parser.add_argument(
"--reference-root",
action="append",
default=[],
help="Old library root stored in crates; may be supplied more than once",
)
parser.add_argument(
"--verbose", action="store_true", help="Write diagnostic progress to stderr"
)
parser.add_argument("--log-file", help="Write scan progress to a log file")
args = parser.parse_args() args = parser.parse_args()
serato = Path(args.serato) config = ScanConfig.build(
music = Path(args.music) serato=Path(args.serato),
out = Path(args.out) music=Path(args.music),
report = Path(args.report) out=Path(args.out),
report=Path(args.report),
reference_roots=(Path(root) for root in args.reference_root),
verbose=args.verbose,
log_file=Path(args.log_file) if args.log_file else None,
)
logger = configure_logging(config.verbose, config.log_file)
logger.info("Starting read-only library inspection")
logger.debug("Serato directory: %s", config.serato)
logger.debug("Music directory: %s", config.music)
refs = parse_crates(serato / "Subcrates") crates = load_library_crates(config.serato, config.reference_roots)
disk = scan_audio(music) filesystem = scan_filesystem(config.music)
database_path = config.serato / "database V2"
database = parse_database(database_path) if database_path.is_file() else None
library = Library.from_crates(
crates=crates,
tracks=filesystem.tracks,
broken_symlinks=filesystem.broken_symlinks,
database=database,
)
results = library.reconcile_by_filename()
missing_count = sum(1 for result in results if not result.exists_by_filename)
logger.info("Parsed %d crate references", len(library.references))
logger.info("Scanned %d disk tracks", len(library.tracks))
logger.info("Found %d references missing by filename", missing_count)
disk_names = {t.filename for t in disk} if args.command == "analyze":
health = analyze_health(library)
rows = [] score = f"{health.score:.1f}%" if health.score is not None else "Not assessed"
for ref in refs: logger.info("Calculated library health: %s", score)
rows.append({ print(f"Overall Health: {score}")
"crate": str(ref.source), print(f"Score Basis: {health.score_basis}")
"serato_path": str(ref.path), print(f"Tracks: {health.disk_tracks}")
"filename": ref.filename, print(f"Crate References: {health.total_references}")
"exists_by_filename": ref.filename in disk_names, print(f"References Scored: {health.scored_references}")
}) print(f"Healthy References: {health.healthy_references}")
print(f"Broken References: {health.missing_references}")
missing_count = sum(1 for r in rows if not r["exists_by_filename"]) print(f"Unique Missing Filenames: {health.unique_missing_filenames}")
print(f"Duplicate Filename Groups: {health.duplicate_filename_groups}")
write_csv(rows, out) print(f"Duplicate Files: {health.duplicate_files}")
write_missing_report(rows, report) print(
"Suspected Cloud Conflict Groups: "
print(f"Crate references: {len(refs)}") f"{health.suspected_cloud_conflict_groups}"
print(f"Disk tracks: {len(disk)}") )
print(
"Suspected Cloud Conflict Files: "
f"{health.suspected_cloud_conflict_files}"
)
print(f"Unused Tracks: {health.unused_tracks}")
print(f"Suggested Matches: {health.suggested_matches}")
print(f"Broken Symlinks: {health.broken_symlinks}")
print(f"Static Crates: {health.static_crates}")
print(f"Smart Crates: {health.smart_crates}")
print(f"Smart Crate Containers: {health.smart_crate_containers}")
print(
f"Dynamic References Excluded: {health.dynamic_references_excluded}"
)
print(f"Database Entries: {health.database_entries}")
print(f"Database / Library Matches: {health.database_library_matches}")
print(f"Database Entries Outside Scan: {health.database_unmatched_entries}")
print(f"Missing Tracks in Serato: {health.database_missing_paths}")
print(
"Unique Missing Tracks in Serato: "
f"{health.database_missing_unique_filenames}"
)
print(f"Tracks Missing From Database: {health.tracks_missing_from_database}")
print(f"Duplicate Database Paths: {health.duplicate_database_paths}")
else:
write_csv(results, config.out)
write_missing_report(results, config.report)
logger.info("Wrote CSV and missing-reference reports")
print(f"Crate references: {len(library.references)}")
print(f"Disk tracks: {len(library.tracks)}")
print(f"Missing by filename: {missing_count}") print(f"Missing by filename: {missing_count}")
print(f"CSV: {out}") print(f"CSV: {config.out}")
print(f"Report: {report}") print(f"Report: {config.report}")
if __name__ == "__main__": if __name__ == "__main__":
+37
View File
@@ -0,0 +1,37 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, Optional, Tuple
@dataclass(frozen=True)
class ScanConfig:
"""Read-only paths used for one library scan."""
serato: Path
music: Path
out: Path
report: Path
reference_roots: Tuple[Path, ...] = ()
verbose: bool = False
log_file: Optional[Path] = None
@classmethod
def build(
cls,
serato: Path,
music: Path,
out: Path,
report: Path,
reference_roots: Iterable[Path] = (),
verbose: bool = False,
log_file: Optional[Path] = None,
) -> "ScanConfig":
return cls(
serato,
music,
out,
report,
tuple(reference_roots),
verbose,
log_file,
)
+71 -8
View File
@@ -1,6 +1,8 @@
from pathlib import Path from pathlib import Path
from typing import Iterable, Tuple
from serato_doctor.models import TrackReference from serato_doctor.models.crate import Crate, CrateKind
from serato_doctor.models.reference import TrackReference
def read_crate_text(crate_path: Path) -> str: def read_crate_text(crate_path: Path) -> str:
@@ -20,19 +22,43 @@ def clean_path(raw: str) -> str:
return raw.strip() return raw.strip()
def parse_crate(crate_path: Path) -> list[TrackReference]: DEFAULT_PATH_MARKERS = ("Users/", "Volumes/")
def path_markers(reference_roots: Iterable[Path]) -> Tuple[str, ...]:
configured = tuple(
root.expanduser().as_posix().strip("/") + "/" for root in reference_roots
)
return configured or DEFAULT_PATH_MARKERS
def classify_crate(crate_path: Path) -> CrateKind:
parent_names = {parent.name.casefold() for parent in crate_path.parents}
if "smartcrates" in parent_names:
return CrateKind.SMART
if crate_path.name.casefold() == "compatible by key.crate":
return CrateKind.SMART
if "subcrates" in parent_names:
return CrateKind.STATIC
return CrateKind.UNKNOWN
def load_crate(crate_path: Path, reference_roots: Iterable[Path] = ()) -> Crate:
text = read_crate_text(crate_path) text = read_crate_text(crate_path)
refs = [] refs = []
marker = "Users/djsplice/OneDrive/Jukebox/"
# Serato record markers seen after paths in UTF-16-LE decoded crate data. # Serato record markers seen after paths in UTF-16-LE decoded crate data.
stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"] stop_markers = ["牴k", "otrk", "ptrk", "tvcn", "ovct"]
for marker in path_markers(reference_roots):
for part in text.split(marker)[1:]: for part in text.split(marker)[1:]:
candidate = marker + part candidate = marker + part
stops = [candidate.find(m) for m in stop_markers if candidate.find(m) != -1] stops = [
candidate.find(stop)
for stop in stop_markers
if candidate.find(stop) != -1
]
if not stops: if not stops:
continue continue
@@ -48,11 +74,48 @@ def parse_crate(crate_path: Path) -> list[TrackReference]:
) )
) )
return refs return Crate(
path=crate_path,
references=tuple(refs),
kind=classify_crate(crate_path),
)
def parse_crates(root: Path) -> list[TrackReference]: def parse_crate(
crate_path: Path, reference_roots: Iterable[Path] = ()
) -> list[TrackReference]:
"""Parse references from one crate, preserving the prototype API."""
return list(load_crate(crate_path, reference_roots).references)
def parse_crates(
root: Path, reference_roots: Iterable[Path] = ()
) -> list[TrackReference]:
refs = [] refs = []
reference_roots = tuple(reference_roots)
for crate in root.rglob("*.crate"): for crate in root.rglob("*.crate"):
refs.extend(parse_crate(crate)) refs.extend(parse_crate(crate, reference_roots))
return refs return refs
def load_library_crates(
serato_root: Path, reference_roots: Iterable[Path] = ()
) -> Tuple[Crate, ...]:
reference_roots = tuple(reference_roots)
if not serato_root.is_dir():
return ()
crates = []
folder_patterns = {
"subcrates": ("*.crate",),
"smartcrates": ("*.scrate", "*.crate"),
}
for folder in serato_root.iterdir():
patterns = folder_patterns.get(folder.name.casefold())
if patterns is None or not folder.is_dir():
continue
for pattern in patterns:
for crate_path in folder.rglob(pattern):
crates.append(load_crate(crate_path, reference_roots))
return tuple(sorted(crates, key=lambda crate: str(crate.path)))
+67
View File
@@ -0,0 +1,67 @@
from pathlib import Path
from typing import Dict, Iterator, Optional, Tuple
from serato_doctor.models.database import DatabaseTrack, SeratoDatabase
TEXT_FIELDS = {
b"tsng": "title",
b"tart": "artist",
b"talb": "album",
b"tgen": "genre",
}
def iter_records(data: bytes) -> Iterator[Tuple[bytes, bytes]]:
"""Yield complete big-endian tag-length-value records."""
offset = 0
while offset + 8 <= len(data):
tag = data[offset : offset + 4]
length = int.from_bytes(data[offset + 4 : offset + 8], "big")
payload_start = offset + 8
payload_end = payload_start + length
if payload_end > len(data):
break
yield tag, data[payload_start:payload_end]
offset = payload_end
def decode_text(payload: bytes) -> Optional[str]:
value = payload.decode("utf-16-be", errors="ignore").strip("\x00").strip()
return value or None
def normalize_database_path(value: str) -> Path:
if value.startswith(("Users/", "Volumes/")):
value = "/" + value
return Path(value)
def parse_track(payload: bytes) -> Optional[DatabaseTrack]:
fields: Dict[str, Optional[str]] = {}
path = None
for tag, value in iter_records(payload):
if tag == b"pfil":
decoded_path = decode_text(value)
if decoded_path:
path = normalize_database_path(decoded_path)
elif tag in TEXT_FIELDS:
fields[TEXT_FIELDS[tag]] = decode_text(value)
if path is None:
return None
return DatabaseTrack(path=path, filename=path.name, **fields)
def parse_database(database_path: Path) -> SeratoDatabase:
data = database_path.read_bytes()
version = None
tracks = []
for tag, payload in iter_records(data):
if tag == b"vrsn":
version = decode_text(payload)
elif tag == b"otrk":
track = parse_track(payload)
if track is not None:
tracks.append(track)
return SeratoDatabase(database_path, version, tuple(tracks))
+43
View File
@@ -0,0 +1,43 @@
from collections import defaultdict
from typing import DefaultDict, Iterable, List, Tuple
from serato_doctor.matching import cloud_conflict_name, normalize
from serato_doctor.models.duplicate import DuplicateGroup, DuplicateKind
from serato_doctor.models.track import DiskTrack
def find_duplicate_groups(
tracks: Iterable[DiskTrack],
) -> Tuple[DuplicateGroup, ...]:
"""Find exact-name duplicates and suspected numeric conflict copies."""
# Healthy symlinks preserve legacy Serato paths without consuming another
# copy of the audio, so they are aliases rather than duplicate files.
track_tuple = tuple(track for track in tracks if not track.path.is_symlink())
by_name: DefaultDict[str, List[DiskTrack]] = defaultdict(list)
by_conflict_name: DefaultDict[str, List[DiskTrack]] = defaultdict(list)
for track in track_tuple:
by_name[normalize(track.filename)].append(track)
by_conflict_name[cloud_conflict_name(track.filename)].append(track)
groups = []
for key, matches in by_name.items():
if len(matches) > 1:
groups.append(_group(DuplicateKind.EXACT_NAME, key, matches))
for key, matches in by_conflict_name.items():
distinct_names = {normalize(track.filename) for track in matches}
if len(distinct_names) > 1:
groups.append(_group(DuplicateKind.CLOUD_CONFLICT, key, matches))
return tuple(
sorted(groups, key=lambda group: (group.kind.value, group.comparison_key))
)
def _group(
kind: DuplicateKind, key: str, tracks: Iterable[DiskTrack]
) -> DuplicateGroup:
ordered = tuple(sorted(tracks, key=lambda track: str(track.path)))
return DuplicateGroup(kind, key, ordered)
+33
View File
@@ -0,0 +1,33 @@
"""Targeted content hashing for duplicate candidates."""
import hashlib
from pathlib import Path
from typing import Iterable
def sha256_file(path: Path, chunk_size: int = 1024 * 1024) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(chunk_size), b""):
digest.update(chunk)
return digest.hexdigest()
def compare_audio_files(paths: Iterable[Path]) -> dict:
paths = tuple(paths)
if len(paths) < 2:
raise ValueError("At least two files are required for comparison")
fingerprints = []
for path in paths:
fingerprints.append(
{
"path": str(path),
"size": path.stat().st_size,
"sha256": sha256_file(path),
}
)
identical = len({item["sha256"] for item in fingerprints}) == 1
return {
"status": "identical" if identical else "different",
"files": fingerprints,
}
+115
View File
@@ -0,0 +1,115 @@
from collections import Counter
from serato_doctor.duplicates import find_duplicate_groups
from serato_doctor.matching import MatchingEngine, normalize
from serato_doctor.models.crate import CrateKind
from serato_doctor.models.duplicate import DuplicateKind
from serato_doctor.models.health import HealthReport
from serato_doctor.models.library import Library
def analyze_health(library: Library) -> HealthReport:
"""Calculate defensible health metrics without changing the library."""
dynamic_sources = {
crate.path for crate in library.crates if crate.kind is CrateKind.SMART
}
results = tuple(
result
for result in library.reconcile_by_filename()
if result.reference.source not in dynamic_sources
)
missing = [result for result in results if not result.exists_by_filename]
healthy_count = len(results) - len(missing)
score = (
round(healthy_count / len(results) * 100, 1) if results else None
)
duplicate_groups = find_duplicate_groups(library.tracks)
exact_duplicates = [
group
for group in duplicate_groups
if group.kind is DuplicateKind.EXACT_NAME
]
cloud_conflicts = [
group
for group in duplicate_groups
if group.kind is DuplicateKind.CLOUD_CONFLICT
]
referenced_names = {reference.filename for reference in library.references}
unused_count = sum(
1 for track in library.tracks if track.filename not in referenced_names
)
matcher = MatchingEngine(library.tracks)
suggested_count = sum(
bool(matcher.candidates_for(result.reference)) for result in missing
)
database_tracks = library.database.tracks if library.database else ()
database_names = {normalize(track.filename) for track in database_tracks}
library_names = {normalize(track.filename) for track in library.tracks}
database_path_counts = Counter(
normalize(str(track.path)) for track in database_tracks
)
missing_database_tracks = [
track for track in database_tracks if not track.path.exists()
]
return HealthReport(
score=score,
total_references=len(library.references),
scored_references=len(results),
healthy_references=healthy_count,
missing_references=len(missing),
unique_missing_filenames=len(
{result.reference.filename for result in missing}
),
disk_tracks=len(library.tracks),
duplicate_filename_groups=len(exact_duplicates),
duplicate_files=sum(group.extra_files for group in exact_duplicates),
suspected_cloud_conflict_groups=len(cloud_conflicts),
suspected_cloud_conflict_files=sum(
group.extra_files for group in cloud_conflicts
),
unused_tracks=unused_count,
suggested_matches=suggested_count,
broken_symlinks=len(library.broken_symlinks),
static_crates=sum(
crate.kind is CrateKind.STATIC for crate in library.crates
),
smart_crates=sum(
crate.kind is CrateKind.SMART and crate.is_smart_definition
for crate in library.crates
),
smart_crate_containers=sum(
crate.kind is CrateKind.SMART and not crate.is_smart_definition
for crate in library.crates
),
unknown_crates=sum(
crate.kind is CrateKind.UNKNOWN for crate in library.crates
),
dynamic_references_excluded=sum(
len(crate.references)
for crate in library.crates
if crate.kind is CrateKind.SMART
),
database_present=library.database is not None,
database_entries=len(database_tracks),
database_library_matches=sum(
normalize(track.filename) in library_names for track in database_tracks
),
database_unmatched_entries=sum(
normalize(track.filename) not in library_names for track in database_tracks
),
database_missing_paths=len(missing_database_tracks),
database_missing_unique_filenames=len(
{normalize(track.filename) for track in missing_database_tracks}
),
tracks_missing_from_database=sum(
normalize(track.filename) not in database_names for track in library.tracks
) if library.database else 0,
duplicate_database_paths=sum(
count - 1 for count in database_path_counts.values() if count > 1
),
)
+40
View File
@@ -0,0 +1,40 @@
import logging
from pathlib import Path
from typing import Optional
LOGGER_NAME = "serato_doctor"
LOG_FORMAT = "%(asctime)s %(levelname)s %(message)s"
def configure_logging(
verbose: bool = False, log_file: Optional[Path] = None
) -> logging.Logger:
"""Configure isolated application logging and return the project logger."""
logger = logging.getLogger(LOGGER_NAME)
logger.setLevel(logging.DEBUG)
logger.propagate = False
for handler in logger.handlers[:]:
handler.close()
logger.removeHandler(handler)
formatter = logging.Formatter(LOG_FORMAT)
if verbose:
console = logging.StreamHandler()
console.setLevel(logging.DEBUG)
console.setFormatter(formatter)
logger.addHandler(console)
if log_file is not None:
file_handler = logging.FileHandler(log_file, encoding="utf-8")
file_handler.setLevel(logging.INFO)
file_handler.setFormatter(formatter)
logger.addHandler(file_handler)
if not logger.handlers:
logger.addHandler(logging.NullHandler())
return logger
+86
View File
@@ -0,0 +1,86 @@
import re
import unicodedata
from collections import defaultdict
from pathlib import Path
from typing import DefaultDict, Iterable, List, Tuple
from serato_doctor.models.match import MatchEvidence, TrackMatch
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def normalize(value: str) -> str:
return unicodedata.normalize("NFKC", value).casefold()
def cloud_conflict_name(filename: str) -> str:
"""Remove a trailing numeric cloud-conflict suffix from a filename stem."""
path = Path(filename)
stem = re.sub(r" \d+$", "", path.stem)
return normalize(stem + path.suffix)
def score_candidate(reference: TrackReference, track: DiskTrack) -> TrackMatch:
reference_name = reference.filename
track_name = track.filename
if reference_name == track_name:
filename_points = 60
filename_reason = "Filename is identical"
elif normalize(reference_name) == normalize(track_name):
filename_points = 55
filename_reason = "Filename matches after case and Unicode normalization"
elif cloud_conflict_name(reference_name) == cloud_conflict_name(track_name):
filename_points = 50
filename_reason = "Filename matches after removing a numeric conflict suffix"
else:
filename_points = 0
filename_reason = "Filename does not match"
same_extension = normalize(reference.path.suffix) == normalize(track.suffix)
same_parent = normalize(reference.path.parent.name) == normalize(
track.path.parent.name
)
evidence = (
MatchEvidence(
"filename",
filename_points > 0,
filename_points,
60,
filename_reason,
),
MatchEvidence(
"extension",
same_extension,
10 if same_extension else 0,
10,
"File extension matches" if same_extension else "File extension differs",
),
MatchEvidence(
"parent_folder",
same_parent,
20 if same_parent else 0,
20,
"Parent folder matches" if same_parent else "Parent folder differs",
),
)
return TrackMatch(reference, track, evidence)
class MatchingEngine:
"""Find and rank filename-related disk candidates without modifying files."""
def __init__(self, tracks: Iterable[DiskTrack]):
self._by_conflict_name: DefaultDict[str, List[DiskTrack]] = defaultdict(list)
for track in tracks:
self._by_conflict_name[cloud_conflict_name(track.filename)].append(track)
def candidates_for(self, reference: TrackReference) -> Tuple[TrackMatch, ...]:
candidates = self._by_conflict_name.get(
cloud_conflict_name(reference.filename), []
)
matches = [score_candidate(reference, track) for track in candidates]
return tuple(
sorted(matches, key=lambda match: (-match.score, str(match.track.path)))
)
+27
View File
@@ -0,0 +1,27 @@
from serato_doctor.models.crate import Crate, CrateKind
from serato_doctor.models.database import DatabaseTrack, SeratoDatabase
from serato_doctor.models.duplicate import DuplicateGroup, DuplicateKind
from serato_doctor.models.filesystem import BrokenSymlink, FilesystemScan
from serato_doctor.models.health import HealthReport
from serato_doctor.models.library import Library
from serato_doctor.models.match import MatchEvidence, TrackMatch
from serato_doctor.models.reference import ReferenceResult, TrackReference
from serato_doctor.models.track import DiskTrack
__all__ = [
"Crate",
"CrateKind",
"DatabaseTrack",
"DiskTrack",
"DuplicateGroup",
"DuplicateKind",
"BrokenSymlink",
"FilesystemScan",
"HealthReport",
"Library",
"MatchEvidence",
"ReferenceResult",
"SeratoDatabase",
"TrackMatch",
"TrackReference",
]
+33
View File
@@ -0,0 +1,33 @@
from dataclasses import dataclass
from enum import Enum
from pathlib import Path
from typing import Tuple
from serato_doctor.models.reference import TrackReference
class CrateKind(str, Enum):
STATIC = "static"
SMART = "smart"
UNKNOWN = "unknown"
@dataclass(frozen=True)
class Crate:
"""A Serato crate and the track references parsed from it."""
path: Path
references: Tuple[TrackReference, ...]
kind: CrateKind = CrateKind.UNKNOWN
@property
def hierarchy(self) -> Tuple[str, ...]:
return tuple(self.path.stem.split("≫≫"))
@property
def display_name(self) -> str:
return self.hierarchy[-1]
@property
def is_smart_definition(self) -> bool:
return self.path.suffix.casefold() == ".scrate"
+22
View File
@@ -0,0 +1,22 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Optional, Tuple
@dataclass(frozen=True)
class DatabaseTrack:
"""Read-only metadata extracted from one database V2 track record."""
path: Path
filename: str
title: Optional[str] = None
artist: Optional[str] = None
album: Optional[str] = None
genre: Optional[str] = None
@dataclass(frozen=True)
class SeratoDatabase:
path: Path
version: Optional[str]
tracks: Tuple[DatabaseTrack, ...]
+30
View File
@@ -0,0 +1,30 @@
from dataclasses import dataclass
from enum import Enum
from typing import Tuple
from serato_doctor.models.track import DiskTrack
class DuplicateKind(str, Enum):
EXACT_NAME = "exact_name"
CLOUD_CONFLICT = "cloud_conflict"
@dataclass(frozen=True)
class DuplicateGroup:
"""A deterministic group of files that warrants duplicate review."""
kind: DuplicateKind
comparison_key: str
tracks: Tuple[DiskTrack, ...]
@property
def extra_files(self) -> int:
return max(0, len(self.tracks) - 1)
@property
def display_name(self) -> str:
return min(
(track.filename for track in self.tracks),
key=lambda name: (len(name), name.casefold()),
)
+21
View File
@@ -0,0 +1,21 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Optional, Tuple
from serato_doctor.models.track import DiskTrack
@dataclass(frozen=True)
class BrokenSymlink:
"""A symbolic link whose target cannot be resolved."""
path: Path
target: Optional[Path]
@dataclass(frozen=True)
class FilesystemScan:
"""Read-only findings from one traversal of a music folder."""
tracks: Tuple[DiskTrack, ...]
broken_symlinks: Tuple[BrokenSymlink, ...]
+39
View File
@@ -0,0 +1,39 @@
from dataclasses import dataclass
from typing import Optional
@dataclass(frozen=True)
class HealthReport:
"""Transparent aggregate findings from a read-only library analysis."""
score: Optional[float]
total_references: int
scored_references: int
healthy_references: int
missing_references: int
unique_missing_filenames: int
disk_tracks: int
duplicate_filename_groups: int
duplicate_files: int
suspected_cloud_conflict_groups: int
suspected_cloud_conflict_files: int
unused_tracks: int
suggested_matches: int
broken_symlinks: int
static_crates: int
smart_crates: int
smart_crate_containers: int
unknown_crates: int
dynamic_references_excluded: int
database_present: bool
database_entries: int
database_library_matches: int
database_unmatched_entries: int
database_missing_paths: int
database_missing_unique_filenames: int
tracks_missing_from_database: int
duplicate_database_paths: int
@property
def score_basis(self) -> str:
return "Resolved non-dynamic references / non-dynamic references scored"
+66
View File
@@ -0,0 +1,66 @@
from dataclasses import dataclass
from typing import Iterable, Optional, Tuple
from serato_doctor.models.crate import Crate
from serato_doctor.models.database import SeratoDatabase
from serato_doctor.models.filesystem import BrokenSymlink
from serato_doctor.models.reference import ReferenceResult, TrackReference
from serato_doctor.models.track import DiskTrack
@dataclass(frozen=True)
class Library:
"""The read-only view of crate references and audio found on disk."""
references: Tuple[TrackReference, ...]
tracks: Tuple[DiskTrack, ...]
crates: Tuple[Crate, ...] = ()
broken_symlinks: Tuple[BrokenSymlink, ...] = ()
database: Optional[SeratoDatabase] = None
@classmethod
def build(
cls,
references: Iterable[TrackReference],
tracks: Iterable[DiskTrack],
broken_symlinks: Iterable[BrokenSymlink] = (),
database: Optional[SeratoDatabase] = None,
) -> "Library":
return cls(
tuple(references),
tuple(tracks),
broken_symlinks=tuple(broken_symlinks),
database=database,
)
@classmethod
def from_crates(
cls,
crates: Iterable[Crate],
tracks: Iterable[DiskTrack],
broken_symlinks: Iterable[BrokenSymlink] = (),
database: Optional[SeratoDatabase] = None,
) -> "Library":
crate_tuple = tuple(crates)
references = tuple(
reference
for crate in crate_tuple
for reference in crate.references
)
return cls(
references,
tuple(tracks),
crate_tuple,
tuple(broken_symlinks),
database,
)
def reconcile_by_filename(self) -> Tuple[ReferenceResult, ...]:
disk_names = {track.filename for track in self.tracks}
return tuple(
ReferenceResult(
reference=reference,
exists_by_filename=reference.filename in disk_names,
)
for reference in self.references
)
+39
View File
@@ -0,0 +1,39 @@
from dataclasses import dataclass
from typing import Tuple
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
@dataclass(frozen=True)
class MatchEvidence:
"""One explainable scoring decision for a candidate track."""
field: str
matched: bool
points: int
max_points: int
explanation: str
@dataclass(frozen=True)
class TrackMatch:
"""A ranked candidate backed by explicit, inspectable evidence."""
reference: TrackReference
track: DiskTrack
evidence: Tuple[MatchEvidence, ...]
@property
def score(self) -> int:
return sum(item.points for item in self.evidence)
@property
def max_score(self) -> int:
return sum(item.max_points for item in self.evidence)
@property
def score_percent(self) -> float:
if not self.max_score:
return 0.0
return round(self.score / self.max_score * 100, 1)
+27
View File
@@ -0,0 +1,27 @@
from dataclasses import dataclass
from pathlib import Path
@dataclass(frozen=True)
class TrackReference:
"""A track path referenced by a Serato crate."""
source: Path
path: Path
filename: str
@dataclass(frozen=True)
class ReferenceResult:
"""The filename-level reconciliation result for a crate reference."""
reference: TrackReference
exists_by_filename: bool
def as_row(self) -> dict:
return {
"crate": str(self.reference.source),
"serato_path": str(self.reference.path),
"filename": self.reference.filename,
"exists_by_filename": self.exists_by_filename,
}
@@ -2,15 +2,10 @@ from dataclasses import dataclass
from pathlib import Path from pathlib import Path
@dataclass(frozen=True)
class TrackReference:
source: Path
path: Path
filename: str
@dataclass(frozen=True) @dataclass(frozen=True)
class DiskTrack: class DiskTrack:
"""An audio file discovered on disk."""
path: Path path: Path
filename: str filename: str
size: int size: int
+242
View File
@@ -0,0 +1,242 @@
"""Backup-first duplicate consolidation.
Serato's binary metadata is deliberately not rewritten. Removed duplicate files
are replaced with symbolic links, so every existing path continues to resolve.
"""
import json
import shutil
from dataclasses import dataclass
from datetime import datetime, timezone
from pathlib import Path
from typing import Iterable, Optional, Tuple
from uuid import uuid4
BACKUP_FOLDER = ".serato-doctor-backups"
@dataclass(frozen=True)
class DuplicateRepairPlan:
keeper: Path
replaced: Tuple[Path, ...]
metadata_files: Tuple[Path, ...]
@property
def changes(self) -> int:
return len(self.replaced)
@dataclass(frozen=True)
class RepairReceipt:
backup: Path
keeper: Path
replaced: Tuple[Path, ...]
@dataclass(frozen=True)
class BackupSummary:
path: Path
created_at: str
keeper: Path
replaced: Tuple[Path, ...]
size_bytes: int
restored_at: Optional[str]
@property
def status(self) -> str:
return "restored" if self.restored_at else "ready"
def plan_duplicate_repair(
keeper: Path, duplicates: Iterable[Path], serato_root: Path
) -> DuplicateRepairPlan:
keeper = keeper.expanduser().resolve()
candidates = tuple(path.expanduser().resolve() for path in duplicates)
if keeper not in candidates:
raise ValueError("The file to keep must belong to this duplicate group")
if not keeper.is_file():
raise ValueError(f"The file to keep no longer exists: {keeper}")
replaced = tuple(path for path in candidates if path != keeper)
if not replaced:
raise ValueError("Choose a duplicate group containing at least two files")
if any(not path.is_file() or path.is_symlink() for path in replaced):
raise ValueError("A duplicate changed since the analysis; analyze again")
metadata = tuple(
sorted(
path
for path in serato_root.expanduser().resolve().rglob("*")
if path.is_file()
and BACKUP_FOLDER not in path.parts
and (
path.name == "database V2"
or path.suffix.casefold() in {".crate", ".scrate"}
)
)
)
return DuplicateRepairPlan(keeper, replaced, metadata)
def apply_duplicate_repair(
plan: DuplicateRepairPlan,
serato_root: Path,
backup_limit: Optional[int] = 10,
) -> RepairReceipt:
"""Create a complete rollback snapshot, then replace extras with symlinks."""
return apply_duplicate_repair_batch((plan,), serato_root, backup_limit)
def apply_duplicate_repair_batch(
plans: Iterable[DuplicateRepairPlan],
serato_root: Path,
backup_limit: Optional[int] = 10,
) -> RepairReceipt:
"""Apply several approved duplicate choices as one atomic backup."""
plans = tuple(plans)
if not plans:
raise ValueError("Choose at least one duplicate group")
if backup_limit is not None and backup_limit < 1:
raise ValueError("Backup limit must be at least 1, or unlimited")
replaced_paths = tuple(
path for plan in plans for path in plan.replaced
)
if len(set(replaced_paths)) != len(replaced_paths):
raise ValueError("The same file appears in more than one repair choice")
keepers = {plan.keeper for plan in plans}
if keepers.intersection(replaced_paths):
raise ValueError("A selected keeper cannot be removed by another choice")
backup_root = serato_root.expanduser().resolve() / BACKUP_FOLDER
backup = backup_root / _backup_name()
files_root = backup / "files"
metadata_root = backup / "serato-metadata"
backup.mkdir(parents=True)
entries = []
try:
for plan in plans:
for path in plan.replaced:
destination = files_root / _safe_backup_path(path)
destination.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(path, destination)
entries.append(
{
"original": str(path),
"backup": str(destination),
"keeper": str(plan.keeper),
}
)
for path in plans[0].metadata_files:
relative = path.relative_to(serato_root.expanduser().resolve())
destination = metadata_root / relative
destination.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(path, destination)
manifest = {
"created_at": datetime.now(timezone.utc).isoformat(),
"keeper": str(plans[0].keeper),
"keepers": [str(plan.keeper) for plan in plans],
"choice_count": len(plans),
"replaced": entries,
"strategy": "symlink",
}
(backup / "manifest.json").write_text(
json.dumps(manifest, indent=2), encoding="utf-8"
)
for entry in entries:
path = Path(entry["original"])
path.unlink()
path.symlink_to(Path(entry["keeper"]))
except Exception:
_rollback_entries(entries)
shutil.rmtree(backup, ignore_errors=True)
raise
rotate_backups(backup_root, backup_limit)
return RepairReceipt(backup, plans[0].keeper, replaced_paths)
def restore_backup(backup: Path) -> Tuple[Path, ...]:
manifest_path = backup.expanduser().resolve() / "manifest.json"
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
restored = []
for entry in manifest["replaced"]:
original = Path(entry["original"])
saved = Path(entry["backup"])
if original.exists() and not original.is_symlink():
raise ValueError(f"Restore would overwrite a real file: {original}")
if original.is_symlink():
original.unlink()
original.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(saved, original)
restored.append(original)
manifest["restored_at"] = datetime.now(timezone.utc).isoformat()
manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
return tuple(restored)
def list_backups(serato_root: Path) -> Tuple[BackupSummary, ...]:
backup_root = serato_root.expanduser().resolve() / BACKUP_FOLDER
if not backup_root.is_dir():
return ()
summaries = []
for backup in backup_root.iterdir():
manifest_path = backup / "manifest.json"
if not manifest_path.is_file():
continue
try:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
replaced = tuple(
Path(entry["original"]) for entry in manifest["replaced"]
)
size = sum(
path.stat().st_size
for path in backup.rglob("*")
if path.is_file()
)
summaries.append(
BackupSummary(
path=backup.resolve(),
created_at=manifest["created_at"],
keeper=Path(manifest["keeper"]),
replaced=replaced,
size_bytes=size,
restored_at=manifest.get("restored_at"),
)
)
except (KeyError, OSError, TypeError, json.JSONDecodeError):
continue
return tuple(
sorted(summaries, key=lambda item: item.created_at, reverse=True)
)
def rotate_backups(backup_root: Path, limit: Optional[int]) -> None:
if limit is None or not backup_root.is_dir():
return
backups = sorted(
(path for path in backup_root.iterdir() if (path / "manifest.json").is_file()),
key=lambda path: path.name,
reverse=True,
)
for expired in backups[limit:]:
shutil.rmtree(expired)
def _backup_name() -> str:
stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
return f"{stamp}-{uuid4().hex[:8]}"
def _safe_backup_path(path: Path) -> Path:
anchorless = path.as_posix().lstrip("/").replace(":", "_")
return Path(anchorless)
def _rollback_entries(entries: Iterable[dict]) -> None:
for entry in entries:
original = Path(entry["original"])
saved = Path(entry["backup"])
if original.is_symlink():
original.unlink()
if not original.exists() and saved.is_file():
original.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(saved, original)
+7 -2
View File
@@ -1,9 +1,13 @@
from collections import Counter, defaultdict from collections import Counter, defaultdict
from pathlib import Path from pathlib import Path
from typing import Iterable
import csv import csv
from serato_doctor.models.reference import ReferenceResult
def write_missing_report(rows: list[dict], out: Path) -> None:
def write_missing_report(results: Iterable[ReferenceResult], out: Path) -> None:
rows = [result.as_row() for result in results]
missing = [r for r in rows if not r["exists_by_filename"]] missing = [r for r in rows if not r["exists_by_filename"]]
crate_counts = Counter(r["crate"] for r in missing) crate_counts = Counter(r["crate"] for r in missing)
@@ -32,7 +36,8 @@ def write_missing_report(rows: list[dict], out: Path) -> None:
f.write(f"- {name}\n") f.write(f"- {name}\n")
def write_csv(rows: list[dict], out: Path) -> None: def write_csv(results: Iterable[ReferenceResult], out: Path) -> None:
rows = [result.as_row() for result in results]
with out.open("w", newline="", encoding="utf-8") as f: with out.open("w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter( writer = csv.DictWriter(
f, f,
+21 -3
View File
@@ -1,14 +1,23 @@
from pathlib import Path from pathlib import Path
from serato_doctor.models import DiskTrack from serato_doctor.models.filesystem import BrokenSymlink, FilesystemScan
from serato_doctor.models.track import DiskTrack
AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"} AUDIO_SUFFIXES = {".mp3", ".m4a", ".wav", ".aif", ".aiff", ".flac"}
def scan_audio(folder: Path) -> list[DiskTrack]: def scan_filesystem(folder: Path) -> FilesystemScan:
tracks = [] tracks = []
broken_symlinks = []
for path in folder.rglob("*"): for path in folder.rglob("*"):
if path.is_symlink() and not path.exists():
try:
target = path.readlink()
except OSError:
target = None
broken_symlinks.append(BrokenSymlink(path=path, target=target))
continue
if not path.is_file(): if not path.is_file():
continue continue
if path.suffix.lower() not in AUDIO_SUFFIXES: if path.suffix.lower() not in AUDIO_SUFFIXES:
@@ -28,4 +37,13 @@ def scan_audio(folder: Path) -> list[DiskTrack]:
) )
) )
return tracks return FilesystemScan(
tracks=tuple(tracks),
broken_symlinks=tuple(broken_symlinks),
)
def scan_audio(folder: Path) -> list[DiskTrack]:
"""Scan audio files while preserving the prototype API."""
return list(scan_filesystem(folder).tracks)
+587
View File
@@ -0,0 +1,587 @@
import argparse
import base64
import binascii
import hmac
import json
import mimetypes
import secrets
import subprocess
from collections import Counter
from dataclasses import asdict
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from importlib import resources
from pathlib import Path
from typing import Iterable, Optional
from urllib.parse import parse_qs, quote, urlsplit
from serato_doctor.crate_parser import load_library_crates
from serato_doctor.database_parser import parse_database
from serato_doctor.duplicates import find_duplicate_groups
from serato_doctor.health import analyze_health
from serato_doctor.hashing import compare_audio_files
from serato_doctor.matching import MatchingEngine, normalize
from serato_doctor.models.crate import CrateKind
from serato_doctor.models.duplicate import DuplicateKind
from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_filesystem
from serato_doctor.repair import (
BACKUP_FOLDER,
apply_duplicate_repair,
apply_duplicate_repair_batch,
list_backups,
plan_duplicate_repair,
restore_backup,
)
MAX_REQUEST_BYTES = 64 * 1024
DETAIL_LIMIT = 50
FILE_TOKEN_SECRET = secrets.token_bytes(32)
STATIC_FILES = {
"/": ("index.html", "text/html; charset=utf-8"),
"/app.css": ("app.css", "text/css; charset=utf-8"),
"/recovery.css": ("recovery.css", "text/css; charset=utf-8"),
"/layout-fixes.css": ("layout-fixes.css", "text/css; charset=utf-8"),
"/app.js": ("app.js", "text/javascript; charset=utf-8"),
}
def _display_path(path: Path) -> str:
return str(path)
def _file_token(path: Path) -> str:
encoded = base64.urlsafe_b64encode(str(path.resolve()).encode()).decode()
signature = hmac.digest(FILE_TOKEN_SECRET, encoded.encode(), "sha256").hex()
return f"{encoded}.{signature}"
def _verified_audio(token: str) -> Path:
try:
encoded, signature = token.rsplit(".", 1)
expected = hmac.digest(
FILE_TOKEN_SECRET, encoded.encode(), "sha256"
).hex()
if not hmac.compare_digest(signature, expected):
raise ValueError
path = Path(base64.urlsafe_b64decode(encoded.encode()).decode())
except (binascii.Error, ValueError, UnicodeDecodeError):
raise ValueError("Invalid or expired file preview")
if not path.is_file() or path.suffix.casefold() not in {
".mp3",
".m4a",
".wav",
".aif",
".aiff",
".flac",
}:
raise ValueError("Audio file is no longer available")
return path
def _file_preview(path: Path) -> dict:
token = _file_token(path)
return {
"path": _display_path(path),
"audio_url": f"/api/audio?token={quote(token)}",
"reveal_token": token,
}
def _first_reason(match) -> str:
for item in match.evidence:
if item.matched:
return item.explanation
return "Filename-related candidate"
def diagnostic_details(library: Library, limit: int = DETAIL_LIMIT) -> dict:
dynamic_sources = {
crate.path for crate in library.crates if crate.kind is CrateKind.SMART
}
results = tuple(
result
for result in library.reconcile_by_filename()
if result.reference.source not in dynamic_sources
)
missing = [result for result in results if not result.exists_by_filename]
matcher = MatchingEngine(library.tracks)
suggested_matches = []
for result in missing:
candidates = matcher.candidates_for(result.reference)
if not candidates:
continue
best = candidates[0]
suggested_matches.append(
{
"filename": result.reference.filename,
"crate": _display_path(result.reference.source),
"saved_path": _display_path(result.reference.path),
"candidate": _display_path(best.track.path),
"score": f"{best.score_percent}%",
"reason": _first_reason(best),
}
)
duplicate_groups = find_duplicate_groups(library.tracks)
exact_duplicates = [
group
for group in duplicate_groups
if group.kind is DuplicateKind.EXACT_NAME
]
cloud_conflicts = [
group
for group in duplicate_groups
if group.kind is DuplicateKind.CLOUD_CONFLICT
]
database_tracks = library.database.tracks if library.database else ()
missing_database_tracks = [
track for track in database_tracks if not track.path.exists()
]
database_filename_counts = Counter(
normalize(track.filename) for track in missing_database_tracks
)
referenced_names = {reference.filename for reference in library.references}
unused_tracks = [
track for track in library.tracks if track.filename not in referenced_names
]
return {
"database_missing_tracks": {
"title": "Missing tracks in Serato",
"summary": (
"These are Serato database entries whose saved file location "
"does not currently exist on disk."
),
"total": len(missing_database_tracks),
"items": [
{
"filename": track.filename,
"saved_path": _display_path(track.path),
"artist": track.artist or "Unknown artist",
"title": track.title or track.filename,
"repeated_filename": database_filename_counts[
normalize(track.filename)
]
> 1,
}
for track in missing_database_tracks[:limit]
],
},
"old_crate_references": {
"title": "Old crate references",
"summary": (
"These are regular crate appearances whose exact filename was "
"not found in the selected music folder."
),
"total": len(missing),
"items": [
{
"filename": result.reference.filename,
"crate": _display_path(result.reference.source),
"saved_path": _display_path(result.reference.path),
}
for result in missing[:limit]
],
},
"suggested_matches": {
"title": "Suggested matches",
"summary": (
"These are read-only guesses where Serato Doctor found a "
"filename-related candidate on disk."
),
"total": len(suggested_matches),
"items": suggested_matches[:limit],
},
"duplicate_filenames": {
"title": "Duplicate filenames",
"summary": (
"These groups contain different files with the same cleaned-up "
"filename. Review before making any decisions."
),
"total": len(exact_duplicates),
"items": [
{
"filename": group.display_name,
"files": [_display_path(track.path) for track in group.tracks],
"file_previews": [
_file_preview(track.path) for track in group.tracks
],
}
for group in exact_duplicates[:limit]
],
},
"cloud_conflicts": {
"title": "Possible cloud conflicts",
"summary": (
"These filename families look like cloud sync conflict copies, "
"such as a duplicate ending in a number."
),
"total": len(cloud_conflicts),
"items": [
{
"filename": group.display_name,
"files": [_display_path(track.path) for track in group.tracks],
"file_previews": [
_file_preview(track.path) for track in group.tracks
],
}
for group in cloud_conflicts[:limit]
],
},
"broken_symlinks": {
"title": "Broken shortcuts",
"summary": (
"These symbolic links point somewhere that no longer resolves."
),
"total": len(library.broken_symlinks),
"items": [
{
"path": _display_path(link.path),
"target": _display_path(link.target) if link.target else "Unknown",
}
for link in library.broken_symlinks[:limit]
],
},
"unused_tracks": {
"title": "Unused tracks",
"summary": (
"These scanned files were not referenced by any loaded crate. "
"That does not mean they should be deleted."
),
"total": len(unused_tracks),
"items": [
{"filename": track.filename, "path": _display_path(track.path)}
for track in unused_tracks[:limit]
],
},
}
def analyze_paths(
serato: Path, music: Path, reference_roots: Iterable[Path] = ()
) -> dict:
serato = serato.expanduser()
music = music.expanduser()
reference_roots = tuple(root.expanduser() for root in reference_roots)
if not serato.is_dir():
raise ValueError(f"Serato folder does not exist: {serato}")
if not music.is_dir():
raise ValueError(f"Music folder does not exist: {music}")
crates = load_library_crates(serato, reference_roots)
filesystem = scan_filesystem(music)
database_path = serato / "database V2"
database = parse_database(database_path) if database_path.is_file() else None
library = Library.from_crates(
crates,
filesystem.tracks,
filesystem.broken_symlinks,
database,
)
report = analyze_health(library)
result = asdict(report)
result["score_basis"] = report.score_basis
result["details"] = diagnostic_details(library)
return result
def duplicate_repair(
serato: Path,
music: Path,
keeper: Path,
group_files: Iterable[Path],
backup_limit: Optional[int],
apply: bool = False,
) -> dict:
"""Validate a current duplicate group and preview or apply consolidation."""
serato = serato.expanduser().resolve()
music = music.expanduser().resolve()
keeper = keeper.expanduser().resolve()
requested = tuple(path.expanduser().resolve() for path in group_files)
if not serato.is_dir() or not music.is_dir():
raise ValueError("Analyze the library again before repairing duplicates")
tracks = scan_filesystem(music).tracks
groups = find_duplicate_groups(tracks)
valid_groups = [
{track.path.resolve() for track in group.tracks} for group in groups
]
if set(requested) not in valid_groups:
raise ValueError("This duplicate group changed; analyze the library again")
plan = plan_duplicate_repair(keeper, requested, serato)
result = {
"keeper": str(plan.keeper),
"replaced": [str(path) for path in plan.replaced],
"metadata_backups": len(plan.metadata_files),
"strategy": "shortcut",
"database_v2_modified": False,
}
if apply:
receipt = apply_duplicate_repair(plan, serato, backup_limit)
result.update({"applied": True, "backup": str(receipt.backup)})
else:
result["applied"] = False
return result
def duplicate_repair_batch(
serato: Path,
music: Path,
choices: Iterable[dict],
backup_limit: Optional[int],
apply: bool = False,
) -> dict:
"""Validate and preview or apply several keeper choices together."""
serato = serato.expanduser().resolve()
music = music.expanduser().resolve()
if not serato.is_dir() or not music.is_dir():
raise ValueError("Analyze the library again before repairing duplicates")
groups = find_duplicate_groups(scan_filesystem(music).tracks)
valid_groups = [
{track.path.resolve() for track in group.tracks} for group in groups
]
plans = []
selected_groups = set()
for choice in choices:
requested = tuple(Path(value).expanduser().resolve() for value in choice["group_files"])
group_key = frozenset(requested)
if set(requested) not in valid_groups:
raise ValueError("A duplicate group changed; analyze the library again")
if group_key in selected_groups:
raise ValueError("A duplicate group was selected more than once")
selected_groups.add(group_key)
plans.append(
plan_duplicate_repair(Path(choice["keeper"]), requested, serato)
)
if not plans:
raise ValueError("Choose at least one duplicate group")
replaced = [str(path) for plan in plans for path in plan.replaced]
comparisons = [
compare_audio_files((plan.keeper,) + plan.replaced) for plan in plans
]
result = {
"choice_count": len(plans),
"replaced": replaced,
"decisions": [
{
"keeper": str(plan.keeper),
"replaced": [str(path) for path in plan.replaced],
"hash_status": comparison["status"],
}
for plan, comparison in zip(plans, comparisons)
],
"metadata_backups": len(plans[0].metadata_files),
"strategy": "shortcut",
"database_v2_modified": False,
}
if apply:
receipt = apply_duplicate_repair_batch(plans, serato, backup_limit)
result.update({"applied": True, "backup": str(receipt.backup)})
else:
result["applied"] = False
return result
def backup_history(serato: Path) -> dict:
serato = serato.expanduser().resolve()
if not serato.is_dir():
raise ValueError(f"Serato folder does not exist: {serato}")
backups = list_backups(serato)
return {
"total": len(backups),
"size_bytes": sum(backup.size_bytes for backup in backups),
"items": [
{
"path": str(backup.path),
"created_at": backup.created_at,
"keeper": str(backup.keeper),
"replaced": [str(path) for path in backup.replaced],
"size_bytes": backup.size_bytes,
"status": backup.status,
"restored_at": backup.restored_at,
}
for backup in backups
],
}
class SeratoDoctorHandler(BaseHTTPRequestHandler):
def do_GET(self) -> None:
request = urlsplit(self.path)
if request.path == "/api/audio":
try:
token = parse_qs(request.query)["token"][0]
self._audio_response(_verified_audio(token))
except (BrokenPipeError, ConnectionResetError):
return
except (KeyError, IndexError, OSError, ValueError) as error:
self._json_response(404, {"error": str(error)})
return
asset = STATIC_FILES.get(request.path)
if asset is None:
self._json_response(404, {"error": "Not found"})
return
filename, content_type = asset
content = (
resources.files("serato_doctor.webui")
.joinpath(filename)
.read_bytes()
)
self.send_response(200)
self.send_header("Content-Type", content_type)
self.send_header("Cache-Control", "no-store")
self.send_header("Content-Length", str(len(content)))
self.end_headers()
self.wfile.write(content)
def do_POST(self) -> None:
allowed = {
"/api/analyze",
"/api/duplicates/preview",
"/api/duplicates/apply",
"/api/duplicates/batch/preview",
"/api/duplicates/batch/apply",
"/api/duplicates/hash",
"/api/backups/restore",
"/api/backups",
"/api/reveal",
}
if self.path not in allowed:
self._json_response(404, {"error": "Not found"})
return
try:
length = int(self.headers.get("Content-Length", "0"))
if length <= 0 or length > MAX_REQUEST_BYTES:
raise ValueError("Invalid request size")
payload = json.loads(self.rfile.read(length))
if not isinstance(payload, dict):
raise ValueError("Request body must be a JSON object")
if self.path == "/api/analyze":
roots = [Path(value) for value in payload.get("reference_roots", [])]
result = analyze_paths(
Path(payload["serato"]), Path(payload["music"]), roots
)
elif self.path == "/api/reveal":
path = _verified_audio(payload["token"])
subprocess.run(["open", "-R", str(path)], check=True)
result = {"revealed": str(path)}
elif self.path == "/api/backups":
result = backup_history(Path(payload["serato"]))
elif self.path == "/api/backups/restore":
serato = Path(payload["serato"]).expanduser().resolve()
backup = Path(payload["backup"]).expanduser().resolve()
backup_root = (serato / BACKUP_FOLDER).resolve()
if backup.parent != backup_root:
raise ValueError("That backup does not belong to this library")
restored = restore_backup(backup)
result = {"restored": [str(path) for path in restored]}
elif self.path == "/api/duplicates/hash":
tokens = payload["tokens"]
if not isinstance(tokens, list) or not 2 <= len(tokens) <= 10:
raise ValueError("Compare between 2 and 10 audio files")
comparison = compare_audio_files(
_verified_audio(token) for token in tokens
)
result = {
"status": comparison["status"],
"files": [
{
"path": item["path"],
"size": item["size"],
"fingerprint": item["sha256"][:12],
}
for item in comparison["files"]
],
}
elif self.path.startswith("/api/duplicates/batch/"):
raw_limit = payload.get("backup_limit", 10)
backup_limit = None if raw_limit is None else int(raw_limit)
result = duplicate_repair_batch(
Path(payload["serato"]),
Path(payload["music"]),
payload["choices"],
backup_limit,
apply=self.path.endswith("/apply"),
)
else:
raw_limit = payload.get("backup_limit", 10)
backup_limit = None if raw_limit is None else int(raw_limit)
result = duplicate_repair(
Path(payload["serato"]),
Path(payload["music"]),
Path(payload["keeper"]),
(Path(value) for value in payload["group_files"]),
backup_limit,
apply=self.path.endswith("/apply"),
)
except (
KeyError,
TypeError,
OSError,
json.JSONDecodeError,
ValueError,
) as error:
self._json_response(400, {"error": str(error)})
return
self._json_response(200, result)
def _audio_response(self, path: Path) -> None:
size = path.stat().st_size
start, end = 0, size - 1
status = 200
range_header = self.headers.get("Range")
if range_header and range_header.startswith("bytes="):
raw_start, _, raw_end = range_header[6:].partition("-")
start = int(raw_start or 0)
end = min(int(raw_end) if raw_end else end, end)
if start < 0 or start > end:
raise ValueError("Invalid audio range")
status = 206
length = end - start + 1
self.send_response(status)
self.send_header(
"Content-Type", mimetypes.guess_type(path.name)[0] or "audio/mpeg"
)
self.send_header("Accept-Ranges", "bytes")
self.send_header("Content-Length", str(length))
if status == 206:
self.send_header("Content-Range", f"bytes {start}-{end}/{size}")
self.end_headers()
with path.open("rb") as audio:
audio.seek(start)
remaining = length
while remaining:
chunk = audio.read(min(64 * 1024, remaining))
if not chunk:
break
self.wfile.write(chunk)
remaining -= len(chunk)
def _json_response(self, status: int, payload: dict) -> None:
content = json.dumps(payload).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json; charset=utf-8")
self.send_header("Content-Length", str(len(content)))
self.end_headers()
self.wfile.write(content)
def log_message(self, format: str, *args: object) -> None:
return
def main() -> None:
parser = argparse.ArgumentParser(prog="serato-doctor-web")
parser.add_argument("--host", default="127.0.0.1")
parser.add_argument("--port", type=int, default=8765)
args = parser.parse_args()
server = ThreadingHTTPServer((args.host, args.port), SeratoDoctorHandler)
print(f"Serato Doctor web interface: http://{args.host}:{args.port}")
print("Press Ctrl+C to stop.")
try:
server.serve_forever()
except KeyboardInterrupt:
pass
finally:
server.server_close()
+1
View File
@@ -0,0 +1 @@
"""Static assets for the local Serato Doctor web interface."""
File diff suppressed because one or more lines are too long
+398
View File
@@ -0,0 +1,398 @@
const form = document.querySelector('#analysis-form');
const button = document.querySelector('#analyze-button');
const errorBox = document.querySelector('#error-message');
const results = document.querySelector('#dashboard');
const infoButtons = document.querySelectorAll('.info-button');
const drillTriggers = document.querySelectorAll('[data-detail]');
const drilldownTitle = document.querySelector('#drilldown-title');
const drilldownCount = document.querySelector('#drilldown-count');
const drilldownSummary = document.querySelector('#drilldown-summary');
const drilldownList = document.querySelector('#drilldown-list');
const repairPanel = document.querySelector('#duplicate-repair');
const repairChoice = document.querySelector('#repair-choice');
const repairPreview = document.querySelector('#repair-preview');
const repairMessage = document.querySelector('#repair-message');
const previewRepairButton = document.querySelector('#preview-repair');
const applyRepairButton = document.querySelector('#apply-repair');
const restoreRepairButton = document.querySelector('#restore-repair');
const loadBackupsButton = document.querySelector('#load-backups');
const backupSummary = document.querySelector('#backup-summary');
const backupList = document.querySelector('#backup-list');
const audioPreview = document.querySelector('#audio-preview');
const audioPreviewName = document.querySelector('#audio-preview-name');
const audioPlayer = document.querySelector('#audio-player');
const closeAudioPreview = document.querySelector('#close-audio-preview');
const batchPreviewModal = document.querySelector('#batch-preview-modal');
const batchPreviewSummary = document.querySelector('#batch-preview-summary');
const batchPreviewList = document.querySelector('#batch-preview-list');
const batchPreviewSafety = document.querySelector('#batch-preview-safety');
const closeBatchPreview = document.querySelector('#close-batch-preview');
const continueReviewing = document.querySelector('#continue-reviewing');
const acceptPreview = document.querySelector('#accept-preview');
let latestAnalysis = null;
let selectedDuplicateGroup = null;
let previewedRepair = null;
let latestBackup = null;
let reviewState = null;
function expandHome(path) {
return path.trim();
}
function escapeHtml(value) {
return String(value ?? '').replace(/[&<>"']/g, (character) => ({
'&': '&amp;',
'<': '&lt;',
'>': '&gt;',
'"': '&quot;',
"'": '&#039;',
}[character]));
}
function formatBytes(bytes) {
if (!bytes) return '0 B';
const units = ['B', 'KB', 'MB', 'GB', 'TB'];
const unit = Math.min(Math.floor(Math.log(bytes) / Math.log(1024)), units.length - 1);
return `${(bytes / (1024 ** unit)).toFixed(unit ? 1 : 0)} ${units[unit]}`;
}
function formatDate(value) {
const date = new Date(value);
return Number.isNaN(date.valueOf()) ? value : date.toLocaleString([], {dateStyle: 'medium', timeStyle: 'short'});
}
async function loadBackups() {
loadBackupsButton.disabled = true;
backupSummary.textContent = 'Looking for recovery snapshots…';
try {
const response = await fetch('/api/backups', {method: 'POST', headers: {'Content-Type': 'application/json'}, body: JSON.stringify({serato: expandHome(document.querySelector('#serato-path').value)})});
const data = await response.json();
if (!response.ok) throw new Error(data.error || 'Could not load backups');
backupSummary.textContent = `${data.total} backup${data.total === 1 ? '' : 's'} · ${formatBytes(data.size_bytes)} on disk`;
backupList.innerHTML = data.items.length ? data.items.map((backup) => `
<article class="backup-item">
<div><span class="backup-status ${backup.status}">${backup.status === 'restored' ? 'Restored' : 'Ready to restore'}</span><strong>${formatDate(backup.created_at)}</strong><small>Kept: ${escapeHtml(backup.keeper)}</small><small>${backup.replaced.length} original file${backup.replaced.length === 1 ? '' : 's'} · ${formatBytes(backup.size_bytes)}</small></div>
<button type="button" data-restore-backup="${escapeHtml(backup.path)}" ${backup.status === 'restored' ? 'disabled' : ''}>${backup.status === 'restored' ? 'Already restored' : 'Restore'}</button>
</article>
`).join('') : '<div class="empty-detail">No repair backups found for this library yet.</div>';
} catch (error) {
backupSummary.textContent = error.message;
backupList.innerHTML = '';
} finally { loadBackupsButton.disabled = false; }
}
loadBackupsButton.addEventListener('click', loadBackups);
backupList.addEventListener('click', async (event) => {
const button = event.target.closest('[data-restore-backup]');
if (!button || button.disabled) return;
if (!window.confirm('Restore the original duplicate files from this backup? Existing real files will never be overwritten.')) return;
button.disabled = true; button.textContent = 'Restoring…';
try {
const response = await fetch('/api/backups/restore', {method: 'POST', headers: {'Content-Type': 'application/json'}, body: JSON.stringify({serato: expandHome(document.querySelector('#serato-path').value), backup: button.dataset.restoreBackup})});
const result = await response.json();
if (!response.ok) throw new Error(result.error || 'Restore failed');
await loadBackups();
} catch (error) { backupSummary.textContent = error.message; button.disabled = false; button.textContent = 'Restore'; }
});
function detailLines(item) {
if (item.file_previews) {
return item.file_previews.map((file) => `<li class="file-compare-row"><span>${escapeHtml(file.path)}</span><div><button type="button" data-audio-url="${escapeHtml(file.audio_url)}" data-audio-name="${escapeHtml(file.path)}">▶ Play preview</button><button type="button" data-reveal-token="${escapeHtml(file.reveal_token)}">Show in Finder</button></div></li>`).join('');
}
const lines = [];
if (item.artist || item.title) lines.push(`${item.artist || 'Unknown artist'}${item.title || item.filename}`);
if (item.crate) lines.push(`Crate: ${item.crate}`);
if (item.saved_path) lines.push(`Saved path: ${item.saved_path}`);
if (item.candidate) lines.push(`Candidate: ${item.candidate}`);
if (item.path) lines.push(`File: ${item.path}`);
if (item.target) lines.push(`Target: ${item.target}`);
if (item.score || item.reason) lines.push(`${item.score || 'Match'} · ${item.reason || 'Candidate found'}`);
if (item.repeated_filename) lines.push('Same filename appears more than once in Seratos missing list.');
return lines.map((line) => `<li>${escapeHtml(line)}</li>`).join('');
}
function renderDetail(key) {
const detail = latestAnalysis?.details?.[key];
if (!detail) return;
drillTriggers.forEach((trigger) => {
trigger.classList.toggle('selected', trigger.dataset.detail === key);
});
drilldownTitle.textContent = detail.title;
drilldownCount.textContent = `${detail.total ?? 0} found`;
drilldownSummary.textContent = detail.summary;
repairPanel.hidden = true;
selectedDuplicateGroup = null;
previewedRepair = null;
latestBackup = null;
restoreRepairButton.hidden = true;
if (!detail.items?.length) {
drilldownList.innerHTML = '<div class="empty-detail">Nothing to review here. Tiny victory parade, very tasteful.</div>';
return;
}
if (detail.items[0].file_previews) {
reviewState = {key, index: 0, choices: new Map()};
repairPanel.hidden = false;
renderReviewGroup();
return;
}
drilldownList.innerHTML = detail.items.map((item, index) => `
<article class="detail-item">
<strong>${escapeHtml(item.filename || item.path || 'Untitled item')}</strong>
<ul>${detailLines(item)}</ul>
</article>
`).join('');
}
function renderReviewGroup() {
const groups = latestAnalysis.details[reviewState.key].items;
const group = groups[reviewState.index];
const chosen = reviewState.choices.get(reviewState.index);
drilldownCount.textContent = `${reviewState.index + 1} of ${groups.length} · ${reviewState.choices.size} approved`;
drilldownSummary.textContent = 'Listen to each candidate, choose the keeper, and well move to the next group. Skip anything uncertain.';
drilldownList.innerHTML = `
<article class="review-workspace">
<div class="review-heading"><div><span>Comparing now</span><strong>${escapeHtml(group.filename)}</strong></div><div class="review-signals"><span id="hash-confidence" class="hash-confidence checking">Checking file identity</span><span>${reviewState.choices.size} selected</span></div></div>
<div class="review-candidates">${group.file_previews.map((file, index) => `
<section class="review-candidate ${chosen === file.path ? 'winner' : ''}">
<span class="candidate-number">Option ${index + 1}</span>
<strong>${escapeHtml(file.path.split('/').pop())}</strong>
<small>${escapeHtml(file.path)}</small>
<div class="file-actions"><button type="button" data-audio-url="${escapeHtml(file.audio_url)}" data-audio-name="${escapeHtml(file.path)}"> Play preview</button><button type="button" data-reveal-token="${escapeHtml(file.reveal_token)}">Show in Finder</button></div>
<button class="choose-winner" type="button" data-choose-winner="${index}">${chosen === file.path ? '✓ Selected keeper' : 'Keep this one →'}</button>
</section>
`).join('')}</div>
<div class="review-navigation"><button type="button" data-review-previous ${reviewState.index === 0 ? 'disabled' : ''}> Previous</button><button type="button" data-review-skip>Skip for now</button></div>
</article>`;
updateBatchSummary();
loadHashConfidence(group, reviewState.index, reviewState.key);
}
async function loadHashConfidence(group, groupIndex, detailKey) {
const badge = document.querySelector('#hash-confidence');
if (group.hashComparison) {
showHashConfidence(badge, group.hashComparison);
return;
}
try {
const response = await fetch('/api/duplicates/hash', {method: 'POST', headers: {'Content-Type': 'application/json'}, body: JSON.stringify({tokens: group.file_previews.map((file) => file.reveal_token)})});
const result = await response.json();
if (!response.ok) throw new Error(result.error || 'Identity check failed');
group.hashComparison = result;
if (reviewState?.index === groupIndex && reviewState?.key === detailKey) showHashConfidence(document.querySelector('#hash-confidence'), result);
} catch (error) {
if (reviewState?.index === groupIndex && reviewState?.key === detailKey && badge) { badge.className = 'hash-confidence unavailable'; badge.textContent = 'Identity not verified'; }
}
}
function showHashConfidence(badge, comparison) {
if (!badge) return;
badge.className = `hash-confidence ${comparison.status}`;
badge.textContent = comparison.status === 'identical' ? '✓ Byte-for-byte identical' : '⚠ Files differ — listen carefully';
badge.title = comparison.files.map((file) => `${formatBytes(file.size)} · ${file.fingerprint}`).join('\n');
}
function updateBatchSummary() {
if (!reviewState) return;
const groups = latestAnalysis.details[reviewState.key].items;
repairChoice.innerHTML = `<div class="batch-summary"><strong>${reviewState.choices.size} group${reviewState.choices.size === 1 ? '' : 's'} approved</strong><span>${groups.length - reviewState.choices.size} skipped or still awaiting a decision</span></div>`;
previewRepairButton.disabled = reviewState.choices.size === 0;
previewedRepair = null;
applyRepairButton.disabled = true;
repairPreview.hidden = true;
}
function repairPayload() {
if (!reviewState?.choices.size) throw new Error('Choose at least one keeper');
const groups = latestAnalysis.details[reviewState.key].items;
const choices = Array.from(reviewState.choices, ([index, keeper]) => ({keeper, group_files: groups[index].files}));
return {serato: expandHome(document.querySelector('#serato-path').value), music: expandHome(document.querySelector('#music-path').value), choices, backup_limit: document.querySelector('#keep-all-backups').checked ? null : Number(document.querySelector('#backup-limit').value)};
}
async function requestRepair(endpoint) {
const response = await fetch(endpoint, {method: 'POST', headers: {'Content-Type': 'application/json'}, body: JSON.stringify(repairPayload())});
const data = await response.json();
if (!response.ok) throw new Error(data.error || 'Duplicate cleanup failed');
return data;
}
previewRepairButton.addEventListener('click', async () => {
repairMessage.textContent = 'Checking the plan…'; applyRepairButton.disabled = true;
try {
previewedRepair = await requestRepair('/api/duplicates/batch/preview');
batchPreviewSummary.textContent = `${previewedRepair.choice_count} keeper decision(s) · ${previewedRepair.replaced.length} duplicate file(s) consolidated`;
batchPreviewList.innerHTML = previewedRepair.decisions.map((decision, index) => `
<article class="preview-decision"><span>Decision ${index + 1}<em class="hash-confidence ${decision.hash_status}">${decision.hash_status === 'identical' ? 'Identical' : 'Files differ'}</em></span><div class="winner-path"><b>Keep</b><strong>${escapeHtml(decision.keeper.split('/').pop())}</strong><small>${escapeHtml(decision.keeper)}</small></div><div class="replaced-paths"><b>Replace with a shortcut</b>${decision.replaced.map((path) => `<small>${escapeHtml(path)}</small>`).join('')}</div></article>
`).join('');
batchPreviewSafety.textContent = `${previewedRepair.metadata_backups} Serato metadata file(s) and every replaced audio file will be backed up. Database V2 will not be modified.`;
repairPreview.innerHTML = `<strong>Preview approved</strong><p>${previewedRepair.choice_count} keeper decision(s) are ready for one backed-up apply.</p>`;
repairPreview.hidden = false; applyRepairButton.disabled = false;
repairMessage.textContent = 'Preview complete. Nothing has changed yet.';
batchPreviewModal.showModal();
} catch (error) { repairMessage.textContent = error.message; }
});
function dismissBatchPreview() { batchPreviewModal.close(); }
closeBatchPreview.addEventListener('click', dismissBatchPreview);
continueReviewing.addEventListener('click', dismissBatchPreview);
acceptPreview.addEventListener('click', dismissBatchPreview);
applyRepairButton.addEventListener('click', async () => {
if (!previewedRepair) return;
applyRepairButton.disabled = true; repairMessage.textContent = 'Creating the backup before making changes…';
try {
const result = await requestRepair('/api/duplicates/batch/apply');
latestBackup = result.backup;
repairMessage.textContent = `Cleanup complete. Restore backup: ${result.backup}`;
previewRepairButton.disabled = true;
restoreRepairButton.hidden = false;
await loadBackups();
} catch (error) { repairMessage.textContent = error.message; applyRepairButton.disabled = false; }
});
restoreRepairButton.addEventListener('click', async () => {
if (!latestBackup) return;
restoreRepairButton.disabled = true; repairMessage.textContent = 'Restoring the duplicate files…';
try {
const response = await fetch('/api/backups/restore', {method: 'POST', headers: {'Content-Type': 'application/json'}, body: JSON.stringify({serato: expandHome(document.querySelector('#serato-path').value), backup: latestBackup})});
const result = await response.json();
if (!response.ok) throw new Error(result.error || 'Restore failed');
repairMessage.textContent = `Restore complete. ${result.restored.length} original file(s) returned.`;
restoreRepairButton.hidden = true;
await loadBackups();
} catch (error) { repairMessage.textContent = error.message; restoreRepairButton.disabled = false; }
});
async function handleFileAction(event) {
const previewButton = event.target.closest('[data-audio-url]');
const revealButton = event.target.closest('[data-reveal-token]');
if (!previewButton && !revealButton) return false;
event.preventDefault(); event.stopPropagation();
if (previewButton) {
audioPlayer.src = previewButton.dataset.audioUrl;
audioPreviewName.textContent = previewButton.dataset.audioName.split('/').pop();
audioPreview.hidden = false;
try { await audioPlayer.play(); } catch (_) { /* Native controls remain available. */ }
} else {
const original = revealButton.textContent;
revealButton.disabled = true; revealButton.textContent = 'Opening…';
try {
const response = await fetch('/api/reveal', {method: 'POST', headers: {'Content-Type': 'application/json'}, body: JSON.stringify({token: revealButton.dataset.revealToken})});
const result = await response.json();
if (!response.ok) throw new Error(result.error || 'Could not open Finder');
revealButton.textContent = 'Shown in Finder';
} catch (error) { revealButton.textContent = error.message; }
setTimeout(() => { revealButton.disabled = false; revealButton.textContent = original; }, 1800);
}
return true;
}
closeAudioPreview.addEventListener('click', () => { audioPlayer.pause(); audioPlayer.removeAttribute('src'); audioPlayer.load(); audioPreview.hidden = true; });
repairChoice.addEventListener('click', handleFileAction);
drilldownList.addEventListener('click', async (event) => {
if (await handleFileAction(event) || !reviewState) return;
const groups = latestAnalysis.details[reviewState.key].items;
const winner = event.target.closest('[data-choose-winner]');
if (winner) {
const fileIndex = Number(winner.dataset.chooseWinner);
reviewState.choices.set(reviewState.index, groups[reviewState.index].files[fileIndex]);
if (reviewState.index < groups.length - 1) reviewState.index += 1;
renderReviewGroup();
return;
}
if (event.target.closest('[data-review-skip]')) {
reviewState.choices.delete(reviewState.index);
if (reviewState.index < groups.length - 1) reviewState.index += 1;
renderReviewGroup();
return;
}
if (event.target.closest('[data-review-previous]') && reviewState.index > 0) {
reviewState.index -= 1;
renderReviewGroup();
}
});
function render(data) {
latestAnalysis = data;
document.querySelectorAll('[data-field]').forEach((element) => {
const value = data[element.dataset.field];
element.textContent = value ?? '—';
});
const score = data.score;
document.querySelector('#health-score').textContent = score == null ? '—' : `${score}%`;
document.querySelector('#score-ring').style.setProperty('--score', score ?? 0);
document.querySelector('#health-message').textContent = score == null
? 'Not enough data yet'
: score >= 95 ? 'Looking excellent' : score >= 80 ? 'A few things need attention' : 'Review recommended';
document.querySelector('#score-basis').textContent = data.score_basis;
document.querySelector('#analysis-time').textContent = `Completed ${new Date().toLocaleTimeString([], {hour: '2-digit', minute: '2-digit'})}`;
renderDetail(data.database_missing_paths > 0 ? 'database_missing_tracks' : 'old_crate_references');
results.hidden = false;
results.scrollIntoView({behavior: 'smooth', block: 'start'});
}
function closeInfoButtons(except = null) {
infoButtons.forEach((infoButton) => {
if (infoButton !== except) {
infoButton.classList.remove('open');
infoButton.setAttribute('aria-expanded', 'false');
}
});
}
infoButtons.forEach((infoButton) => {
infoButton.addEventListener('click', (event) => {
event.stopPropagation();
const willOpen = !infoButton.classList.contains('open');
closeInfoButtons(infoButton);
infoButton.classList.toggle('open', willOpen);
infoButton.setAttribute('aria-expanded', String(willOpen));
});
});
drillTriggers.forEach((trigger) => {
trigger.addEventListener('click', () => renderDetail(trigger.dataset.detail));
trigger.addEventListener('keydown', (event) => {
if (event.key === 'Enter' || event.key === ' ') {
event.preventDefault();
renderDetail(trigger.dataset.detail);
}
});
});
document.addEventListener('click', () => closeInfoButtons());
document.addEventListener('keydown', (event) => {
if (event.key === 'Escape') closeInfoButtons();
});
form.addEventListener('submit', async (event) => {
event.preventDefault();
errorBox.hidden = true;
button.disabled = true;
button.querySelector('span').textContent = 'Analyzing safely…';
const roots = document.querySelector('#reference-roots').value
.split('\n').map((value) => value.trim()).filter(Boolean);
try {
const response = await fetch('/api/analyze', {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
serato: expandHome(document.querySelector('#serato-path').value),
music: expandHome(document.querySelector('#music-path').value),
reference_roots: roots,
}),
});
const data = await response.json();
if (!response.ok) throw new Error(data.error || 'Analysis failed');
render(data);
} catch (error) {
errorBox.textContent = error.message;
errorBox.hidden = false;
} finally {
button.disabled = false;
button.querySelector('span').textContent = 'Analyze library';
}
});
+135
View File
@@ -0,0 +1,135 @@
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<meta name="color-scheme" content="dark">
<title>Serato Doctor</title>
<link rel="stylesheet" href="/app.css?v=3">
<link rel="stylesheet" href="/recovery.css?v=3">
<link rel="stylesheet" href="/layout-fixes.css?v=3">
</head>
<body>
<div class="ambient ambient-one"></div>
<div class="ambient ambient-two"></div>
<div class="shell">
<aside class="sidebar">
<a class="brand" href="/" aria-label="Serato Doctor home">
<span class="brand-mark">SD</span>
<span><strong>Serato</strong><small>Doctor</small></span>
</a>
<nav aria-label="Primary navigation">
<a class="nav-item active" href="#dashboard"><span></span> Dashboard</a>
<a class="nav-item" href="#scan"><span></span> New analysis</a>
<a class="nav-item" href="#diagnostics"><span></span> Diagnostics</a>
<a class="nav-item" href="#recovery"><span></span> Recovery</a>
</nav>
<div class="safety-card">
<span class="safety-icon"></span>
<div><strong>Protected mode</strong><p>Analysis is read-only. Repairs require a preview and backup.</p></div>
</div>
<div class="sidebar-foot">Local interface · v0.1</div>
</aside>
<main>
<header class="topbar">
<div><p class="eyebrow">Library intelligence</p><h1>Good evening.</h1></div>
<div class="status-pill"><span></span> Local &amp; private</div>
</header>
<section id="scan" class="scan-panel panel">
<div class="panel-copy">
<p class="eyebrow">Start here</p>
<h2>Analyze your library</h2>
<p>Point Serato Doctor at your Serato and music folders. We inspect references, duplicates, smart crates, and symlinks without changing a thing.</p>
</div>
<form id="analysis-form">
<label>Serato folder
<input id="serato-path" name="serato" value="~/Music/_Serato_" required>
</label>
<label>Music folder
<input id="music-path" name="music" value="~/Music/Jukebox" required>
</label>
<label class="wide">Historical reference roots <span>optional · one per line</span>
<textarea id="reference-roots" rows="2" placeholder="/Users/old-user/OneDrive/Jukebox"></textarea>
</label>
<button id="analyze-button" type="submit"><span>Analyze library</span><b></b></button>
</form>
<div id="error-message" class="error" role="alert" hidden></div>
</section>
<section id="dashboard" class="results" aria-live="polite" hidden>
<div class="section-heading"><div><p class="eyebrow">Latest analysis</p><h2>Library health</h2></div><span id="analysis-time"></span></div>
<div class="hero-grid">
<article class="score-card panel">
<button class="info-button" type="button" aria-label="About library health" aria-expanded="false" data-info="Your health score is the percentage of saved, non-smart crate entries whose filenames were found in the selected music folder. Smart crates are left out because Serato rebuilds them from rules.">i</button>
<div class="score-ring" id="score-ring"><div><strong id="health-score"></strong><span>health</span></div></div>
<div><p class="score-label">Reference integrity</p><h3 id="health-message">Ready to analyze</h3><p id="score-basis">We only score evidence we can defend.</p></div>
</article>
<div class="metrics-grid">
<article class="metric panel"><button class="info-button" type="button" aria-label="About tracks" aria-expanded="false" data-info="Audio files found inside the music folder you selected. This is the collection Serato Doctor compared with your crates and database.">i</button><span>Tracks scanned</span><strong data-field="disk_tracks"></strong><small>audio files found</small></article>
<article class="metric panel warning drill-trigger" role="button" tabindex="0" data-detail="database_missing_tracks"><button class="info-button" type="button" aria-label="About missing tracks in Serato" aria-expanded="false" data-info="Tracks in Serato's database whose saved file location no longer exists. This should be close to the orange or unmapped track count you see in Serato.">i</button><span>Missing tracks in Serato</span><strong data-field="database_missing_paths"></strong><small><b data-field="database_missing_unique_filenames"></b> unique filenames · click for list</small></article>
<article class="metric panel drill-trigger" role="button" tabindex="0" data-detail="unused_tracks"><button class="info-button" type="button" aria-label="About unused tracks" aria-expanded="false" data-info="Files in the selected music folder whose filename is not used by any loaded crate. They may still be valid library tracks; this is informational, not a deletion recommendation.">i</button><span>Unused tracks</span><strong data-field="unused_tracks"></strong><small>not referenced by crates · click for list</small></article>
<article class="metric panel drill-trigger" role="button" tabindex="0" data-detail="suggested_matches"><button class="info-button" type="button" aria-label="About suggested matches" aria-expanded="false" data-info="Missing crate entries with a filename-related candidate, such as an added OneDrive conflict number. Suggestions are evidence for review, never automatic repairs.">i</button><span>Suggested matches</span><strong data-field="suggested_matches"></strong><small>explainable candidates · click for list</small></article>
</div>
</div>
<div id="diagnostics" class="diagnostics panel">
<div class="section-heading"><div><p class="eyebrow">Full picture</p><h2>Diagnostics</h2></div><span class="read-only-tag">No changes made</span></div>
<div class="diagnostic-list">
<div><span class="diag-icon violet"></span><p><strong>Crates</strong><small><b data-field="static_crates"></b> regular · <b data-field="smart_crates"></b> smart · <b data-field="smart_crate_containers"></b> dynamic containers</small></p><button class="info-button" type="button" aria-label="About crates" aria-expanded="false" data-info="Regular crates are lists you maintain by hand. Smart crates are rebuilt by Serato from rules, so their generated references are not scored as broken.">i</button></div>
<div class="drill-trigger" role="button" tabindex="0" data-detail="duplicate_filenames"><span class="diag-icon amber"></span><p><strong>Duplicate filenames</strong><small><b data-field="duplicate_filename_groups"></b> exact groups · <b data-field="duplicate_files"></b> extra files · click for list</small></p><button class="info-button" type="button" aria-label="About duplicate filenames" aria-expanded="false" data-info="Different files with the same filename after case and Unicode cleanup. They need review, but matching names alone do not mean either file should be deleted.">i</button></div>
<div class="drill-trigger" role="button" tabindex="0" data-detail="cloud_conflicts"><span class="diag-icon blue"></span><p><strong>Possible cloud conflicts</strong><small><b data-field="suspected_cloud_conflict_groups"></b> groups · <b data-field="suspected_cloud_conflict_files"></b> extra files · click for list</small></p><button class="info-button" type="button" aria-label="About cloud conflicts" aria-expanded="false" data-info="Filename families such as Track.mp3 and Track 2.mp3. OneDrive often creates these during sync conflicts, but numbered song titles can also be legitimate.">i</button></div>
<div class="drill-trigger" role="button" tabindex="0" data-detail="broken_symlinks"><span class="diag-icon red"></span><p><strong>Broken shortcuts</strong><small><b data-field="broken_symlinks"></b> unresolved symbolic links · click for list</small></p><button class="info-button" type="button" aria-label="About broken shortcuts" aria-expanded="false" data-info="Shortcut-style symbolic links whose destination no longer exists. Serato Doctor reports them but never removes or recreates them automatically.">i</button></div>
<div class="drill-trigger" role="button" tabindex="0" data-detail="old_crate_references"><span class="diag-icon violet"></span><p><strong>Old crate references</strong><small><b data-field="missing_references"></b> appearances · <b data-field="unique_missing_filenames"></b> unique filenames · click for list</small></p><button class="info-button" type="button" aria-label="About old crate references" aria-expanded="false" data-info="Saved spots in regular crates whose exact filename was not found in the selected music folder. The same track can appear in several crates, so appearances are higher than unique filenames. These are separate from Serato's unmapped-track count.">i</button></div>
<div><span class="diag-icon amber"></span><p><strong>Database coverage</strong><small><b data-field="database_entries"></b> Serato entries · <b data-field="tracks_missing_from_database"></b> scanned tracks absent</small></p><button class="info-button" type="button" aria-label="About database coverage" aria-expanded="false" data-info="Compares filenames in Serato's database with the selected music folder. A scanned track absent from the database may not have been imported, or may be represented under another filename.">i</button></div>
</div>
</div>
<div id="drilldowns" class="drilldowns panel">
<div class="section-heading"><div><p class="eyebrow">Look closer</p><h2 id="drilldown-title">Choose a diagnostic</h2></div><span id="drilldown-count">Read-only examples</span></div>
<p id="drilldown-summary">Click a metric above to see example files and saved paths behind that number.</p>
<div id="drilldown-list" class="detail-list"></div>
</div>
<div id="duplicate-repair" class="repair-panel panel" hidden>
<div class="section-heading"><div><p class="eyebrow">Batch cleanup</p><h2>Review all selected changes</h2></div><span class="repair-tag">One backup · one apply</span></div>
<p>Approved choices are combined into one plan. Skipped groups remain untouched, and Seratos database V2 is never edited.</p>
<div id="repair-choice" class="repair-choice"></div>
<div class="backup-options">
<label>Backups to keep <input id="backup-limit" type="number" min="1" value="10"></label>
<label class="check-label"><input id="keep-all-backups" type="checkbox"> Keep every backup</label>
</div>
<div id="repair-preview" class="repair-preview" hidden></div>
<div class="repair-actions">
<button id="preview-repair" type="button">Preview all selected changes</button>
<button id="apply-repair" class="danger-action" type="button" disabled>Apply all selected changes</button>
<button id="restore-repair" type="button" hidden>Restore this backup</button>
</div>
<div id="repair-message" class="repair-message" role="status"></div>
</div>
</section>
<section id="recovery" class="recovery-panel panel">
<div class="section-heading"><div><p class="eyebrow">Safety net</p><h2>Backup recovery</h2></div><button id="load-backups" type="button">Check backups</button></div>
<p>Repair backups appear here automatically before any duplicate is consolidated. Review saved snapshots, disk usage, and restore history.</p>
<div id="backup-summary" class="backup-summary">Analyze your library, then choose a duplicate to preview its automatic backup.</div>
<div id="backup-list" class="backup-list"></div>
</section>
<div id="audio-preview" class="audio-preview" hidden>
<div><span>Now previewing</span><strong id="audio-preview-name"></strong></div>
<audio id="audio-player" controls preload="metadata"></audio>
<button id="close-audio-preview" type="button" aria-label="Close audio preview">×</button>
</div>
<dialog id="batch-preview-modal" class="batch-preview-modal">
<div class="modal-heading"><div><p class="eyebrow">Final dry-run</p><h2>Heres exactly what will happen</h2></div><button id="close-batch-preview" type="button" aria-label="Close preview">×</button></div>
<p id="batch-preview-summary"></p>
<div id="batch-preview-list" class="batch-preview-list"></div>
<div class="modal-safety"><span></span><p><strong>Backup happens first</strong><small id="batch-preview-safety"></small></p></div>
<div class="modal-actions"><button id="continue-reviewing" type="button">Close and continue reviewing</button><button id="accept-preview" type="button">Looks right</button></div>
</dialog>
</main>
</div>
<script src="/app.js?v=6" defer></script>
</body>
</html>
+447
View File
@@ -0,0 +1,447 @@
.recovery-panel {
margin-top: 42px;
}
@media (max-width: 680px) {
.recovery-panel {
margin-top: 28px;
}
}
.review-workspace {
padding: clamp(16px, 3vw, 24px);
border: 1px solid rgba(155, 135, 245, .18);
border-radius: 16px;
background: rgba(255, 255, 255, .02);
}
.review-heading,
.review-navigation {
display: flex;
align-items: center;
justify-content: space-between;
gap: 16px;
}
.review-heading span,
.batch-summary span {
color: var(--muted);
font-size: 10px;
}
.review-heading strong {
display: block;
margin-top: 5px;
font-size: 14px;
}
.review-signals {
display: flex;
align-items: flex-end;
gap: 7px;
flex-direction: column;
}
.hash-confidence {
display: inline-block;
padding: 5px 8px;
border-radius: 999px;
font-size: 9px!important;
font-style: normal;
font-weight: 750;
letter-spacing: .02em!important;
text-transform: none!important;
}
.hash-confidence.checking,
.hash-confidence.unavailable {
color: var(--muted);
background: rgba(255, 255, 255, .05);
}
.hash-confidence.identical {
color: #8ce4b8;
background: rgba(84, 212, 154, .11);
}
.hash-confidence.different {
color: var(--amber);
background: rgba(245, 185, 76, .1);
}
.preview-decision > span .hash-confidence {
display: block;
width: max-content;
margin-top: 8px;
}
.review-candidates {
display: grid;
grid-template-columns: repeat(2, minmax(0, 1fr));
gap: 12px;
margin: 18px 0;
}
.review-candidate {
display: flex;
flex-direction: column;
gap: 9px;
min-width: 0;
padding: 16px;
border: 1px solid var(--line);
border-radius: 14px;
background: rgba(6, 8, 12, .28);
}
.review-candidate.winner {
border-color: rgba(84, 212, 154, .52);
background: rgba(84, 212, 154, .06);
}
.candidate-number {
color: var(--violet);
font-size: 9px;
font-weight: 750;
text-transform: uppercase;
letter-spacing: .1em;
}
.review-candidate > strong,
.review-candidate > small {
overflow-wrap: anywhere;
}
.review-candidate > strong {
font-size: 12px;
}
.review-candidate > small {
flex: 1;
color: var(--muted);
font-size: 9px;
line-height: 1.5;
}
.choose-winner {
margin-top: 4px;
border: 0;
border-radius: 10px;
padding: 10px 12px;
background: linear-gradient(135deg, #907be9, #6e59cf);
color: white;
font: 750 11px/1 inherit;
cursor: pointer;
}
.winner .choose-winner {
background: rgba(84, 212, 154, .18);
color: #8ce4b8;
}
.review-navigation button {
border: 1px solid var(--line);
border-radius: 9px;
padding: 8px 11px;
background: rgba(255, 255, 255, .04);
color: var(--muted);
font: 700 10px/1 inherit;
cursor: pointer;
}
.review-navigation button:disabled {
opacity: .35;
}
.batch-summary {
display: flex;
align-items: center;
justify-content: space-between;
gap: 12px;
padding: 14px;
border: 1px solid rgba(102, 217, 232, .2);
border-radius: 12px;
background: rgba(102, 217, 232, .05);
}
.batch-summary strong {
font-size: 12px;
}
@media (max-width: 760px) {
.review-candidates {
grid-template-columns: 1fr;
}
.review-heading,
.batch-summary {
align-items: flex-start;
flex-direction: column;
}
}
.batch-preview-modal {
width: min(820px, calc(100vw - 32px));
max-height: min(82vh, 760px);
padding: 24px;
overflow: auto;
border: 1px solid rgba(155, 135, 245, .3);
border-radius: 20px;
background: #161923;
color: var(--text);
box-shadow: 0 30px 100px rgba(0, 0, 0, .65);
}
.batch-preview-modal::backdrop {
background: rgba(4, 5, 8, .78);
backdrop-filter: blur(7px);
}
.modal-heading,
.modal-actions,
.modal-safety {
display: flex;
align-items: center;
justify-content: space-between;
gap: 14px;
}
.modal-heading h2 {
margin: 0;
font-size: 23px;
letter-spacing: -.03em;
}
.modal-heading > button {
border: 0;
background: transparent;
color: var(--muted);
font-size: 28px;
cursor: pointer;
}
#batch-preview-summary {
color: var(--muted);
font-size: 12px;
}
.batch-preview-list {
display: grid;
gap: 10px;
max-height: 390px;
margin: 18px 0;
padding-right: 4px;
overflow: auto;
}
.preview-decision {
display: grid;
grid-template-columns: 80px 1fr 1fr;
gap: 13px;
padding: 14px;
border: 1px solid var(--line);
border-radius: 13px;
background: rgba(255, 255, 255, .025);
}
.preview-decision > span,
.preview-decision b {
color: var(--muted);
font-size: 9px;
text-transform: uppercase;
letter-spacing: .08em;
}
.winner-path,
.replaced-paths {
min-width: 0;
}
.winner-path strong,
.winner-path small,
.replaced-paths small {
display: block;
margin-top: 5px;
overflow-wrap: anywhere;
}
.winner-path strong {
color: #8ce4b8;
font-size: 11px;
}
.winner-path small,
.replaced-paths small {
color: var(--muted);
font-size: 9px;
line-height: 1.45;
}
.modal-safety {
justify-content: flex-start;
padding: 13px;
border: 1px solid rgba(102, 217, 232, .2);
border-radius: 12px;
background: rgba(102, 217, 232, .05);
}
.modal-safety > span {
color: var(--cyan);
}
.modal-safety p,
.modal-safety strong,
.modal-safety small {
display: block;
margin: 0;
}
.modal-safety strong {
font-size: 11px;
}
.modal-safety small {
margin-top: 4px;
color: var(--muted);
font-size: 9px;
line-height: 1.45;
}
.modal-actions {
justify-content: flex-end;
margin-top: 16px;
}
.modal-actions button {
border: 1px solid var(--line);
border-radius: 10px;
padding: 10px 13px;
background: rgba(255, 255, 255, .04);
color: var(--text);
font: 700 11px/1 inherit;
cursor: pointer;
}
#accept-preview {
border-color: transparent;
background: #6f5bd0;
}
@media (max-width: 620px) {
.preview-decision {
grid-template-columns: 1fr;
}
.modal-actions {
align-items: stretch;
flex-direction: column-reverse;
}
}
.file-compare-row {
display: grid;
gap: 8px;
padding: 9px 0;
border-top: 1px solid rgba(255, 255, 255, .05);
}
.file-compare-row:first-child {
border-top: 0;
}
.file-compare-row div,
.file-actions {
display: flex;
flex-wrap: wrap;
gap: 7px;
}
.file-compare-row button,
.file-actions button {
border: 1px solid rgba(155, 135, 245, .25);
border-radius: 8px;
padding: 7px 9px;
background: rgba(155, 135, 245, .08);
color: #d8d2f6;
font: 700 10px/1 inherit;
cursor: pointer;
}
.keeper-option {
justify-content: space-between;
}
.keeper-option > label {
display: flex;
align-items: center;
gap: 12px;
min-width: 0;
cursor: pointer;
}
.audio-preview {
position: fixed;
right: 24px;
bottom: 20px;
display: grid;
grid-template-columns: minmax(150px, .7fr) minmax(240px, 1.3fr) auto;
align-items: center;
gap: 16px;
width: min(720px, calc(100vw - 48px));
padding: 13px 14px;
border: 1px solid rgba(102, 217, 232, .3);
border-radius: 14px;
background: rgba(16, 21, 29, .96);
box-shadow: 0 16px 48px rgba(0, 0, 0, .45);
z-index: 5;
}
.audio-preview[hidden] {
display: none;
}
.audio-preview span,
.audio-preview strong {
display: block;
}
.audio-preview span {
color: var(--cyan);
font-size: 9px;
text-transform: uppercase;
letter-spacing: .1em;
}
.audio-preview strong {
max-width: 360px;
margin-top: 4px;
overflow: hidden;
color: var(--text);
font-size: 11px;
text-overflow: ellipsis;
white-space: nowrap;
}
.audio-preview audio {
width: 100%;
height: 34px;
}
.audio-preview > button {
border: 0;
background: transparent;
color: var(--muted);
font-size: 22px;
cursor: pointer;
}
@media (max-width: 680px) {
.keeper-option,
.audio-preview {
align-items: stretch;
grid-template-columns: 1fr;
}
.keeper-option {
flex-direction: column;
}
}
+1
View File
@@ -0,0 +1 @@
.recovery-panel{margin-top:18px;padding:25px}.recovery-panel>p{color:var(--muted);font-size:12px;line-height:1.6}.recovery-panel .section-heading button{border:1px solid rgba(155,135,245,.35);border-radius:10px;padding:9px 12px;background:rgba(155,135,245,.12);color:#ddd8fa;font:700 11px inherit;cursor:pointer}.recovery-panel button:disabled{opacity:.5;cursor:not-allowed}.backup-summary{padding:11px 13px;border-radius:11px;background:rgba(255,255,255,.03);color:#bfc4cf;font-size:11px;margin:15px 0 10px}.backup-list{display:grid;gap:9px}.backup-item{display:flex;align-items:center;justify-content:space-between;gap:18px;padding:14px;border:1px solid var(--line);border-radius:13px;background:rgba(255,255,255,.025)}.backup-item>div{min-width:0}.backup-item strong,.backup-item small{display:block}.backup-item strong{font-size:12px;margin:7px 0}.backup-item small{font-size:10px;color:var(--muted);line-height:1.5;overflow-wrap:anywhere}.backup-item>button{flex:0 0 auto;border:0;border-radius:10px;padding:10px 13px;background:#6f5bd0;color:#fff;font:700 11px inherit;cursor:pointer}.backup-status{display:inline-block;padding:4px 7px;border-radius:999px;font-size:9px;text-transform:uppercase;letter-spacing:.08em;color:var(--cyan);background:rgba(102,217,232,.1)}.backup-status.restored{color:#aeb4c2;background:rgba(255,255,255,.06)}@media(max-width:520px){.backup-item{align-items:stretch;flex-direction:column}.backup-item>button{width:100%}}
+131
View File
@@ -0,0 +1,131 @@
import csv
import sys
from serato_doctor.cli import main
def test_cli_writes_reports_and_prints_counts(tmp_path, monkeypatch, capsys):
serato = tmp_path / "serato"
subcrates = serato / "Subcrates"
music = tmp_path / "music"
subcrates.mkdir(parents=True)
music.mkdir()
crate_text = (
"Users/sample-user/OneDrive/Jukebox/Found.mp漳牴k"
"Users/sample-user/OneDrive/Jukebox/Missing.mp漳牴k"
)
(subcrates / "Test.crate").write_bytes(crate_text.encode("utf-16-le"))
(music / "Found.mp3").write_bytes(b"synthetic audio")
csv_path = tmp_path / "scan.csv"
report_path = tmp_path / "report.txt"
monkeypatch.setattr(
sys,
"argv",
[
"serato-doctor",
"--serato",
str(serato),
"--music",
str(music),
"--out",
str(csv_path),
"--report",
str(report_path),
],
)
main()
captured = capsys.readouterr()
output = captured.out
assert captured.err == ""
assert "Crate references: 2" in output
assert "Disk tracks: 1" in output
assert "Missing by filename: 1" in output
assert f"CSV: {csv_path}" in output
assert f"Report: {report_path}" in output
with csv_path.open(newline="", encoding="utf-8") as csv_file:
rows = list(csv.DictReader(csv_file))
assert [row["exists_by_filename"] for row in rows] == ["True", "False"]
assert "Total missing references: 1" in report_path.read_text(encoding="utf-8")
def test_cli_accepts_old_reference_root(tmp_path, monkeypatch, capsys):
serato = tmp_path / "serato"
subcrates = serato / "Subcrates"
music = tmp_path / "music"
subcrates.mkdir(parents=True)
music.mkdir()
(subcrates / "Test.crate").write_bytes(
"Archive/Jukebox/Found.mp3otrk".encode("utf-16-le")
)
(music / "Found.mp3").write_bytes(b"synthetic audio")
monkeypatch.setattr(
sys,
"argv",
[
"serato-doctor",
"--serato",
str(serato),
"--music",
str(music),
"--out",
str(tmp_path / "scan.csv"),
"--report",
str(tmp_path / "report.txt"),
"--reference-root",
"/Archive/Jukebox",
],
)
main()
output = capsys.readouterr().out
assert "Crate references: 1" in output
assert "Missing by filename: 0" in output
def test_analyze_prints_health_without_writing_reports(tmp_path, monkeypatch, capsys):
serato = tmp_path / "serato"
subcrates = serato / "Subcrates"
music = tmp_path / "music"
subcrates.mkdir(parents=True)
music.mkdir()
crate_text = (
"Users/sample-user/Jukebox/Found.mp3otrk"
"Users/sample-user/Jukebox/Missing.mp3otrk"
)
(subcrates / "Test.crate").write_bytes(crate_text.encode("utf-16-le"))
(music / "Found.mp3").write_bytes(b"synthetic audio")
csv_path = tmp_path / "scan.csv"
report_path = tmp_path / "report.txt"
monkeypatch.setattr(
sys,
"argv",
[
"serato-doctor",
"analyze",
"--serato",
str(serato),
"--music",
str(music),
"--out",
str(csv_path),
"--report",
str(report_path),
],
)
main()
output = capsys.readouterr().out
assert "Overall Health: 50.0%" in output
assert "Tracks: 1" in output
assert "Crate References: 2" in output
assert "References Scored: 2" in output
assert "Healthy References: 1" in output
assert "Broken References: 1" in output
assert "Unused Tracks: 0" in output
assert not csv_path.exists()
assert not report_path.exists()
+17
View File
@@ -0,0 +1,17 @@
from pathlib import Path
from serato_doctor.config import ScanConfig
def test_scan_config_freezes_reference_roots():
roots = (Path(root) for root in ["/old/one", "/old/two"])
config = ScanConfig.build(
serato=Path("serato"),
music=Path("music"),
out=Path("scan.csv"),
report=Path("report.txt"),
reference_roots=roots,
)
assert config.reference_roots == (Path("/old/one"), Path("/old/two"))
+136
View File
@@ -0,0 +1,136 @@
from pathlib import Path
import pytest
from serato_doctor.crate_parser import (
classify_crate,
clean_path,
load_crate,
load_library_crates,
parse_crate,
parse_crates,
path_markers,
)
from serato_doctor.models.crate import CrateKind
@pytest.mark.parametrize(
("artifact", "expected"),
[
("song.mp漳", "song.mp3"),
("song.m4愠", "song.m4a"),
("song.wa瘠", "song.wav"),
("song.ai映", "song.aif"),
],
)
def test_clean_path_repairs_known_extension_artifacts(artifact, expected):
assert clean_path(artifact) == expected
def test_parse_crate_extracts_references(tmp_path):
crate_path = tmp_path / "House.crate"
crate_text = (
"header"
"Users/sample-user/OneDrive/Jukebox/House/First.mp漳牴k"
"metadata"
"Users/sample-user/OneDrive/Jukebox/House/Second.m4愠otrk"
)
crate_path.write_bytes(crate_text.encode("utf-16-le"))
crate = load_crate(crate_path)
assert crate.path == crate_path
assert [reference.filename for reference in crate.references] == [
"First.mp3",
"Second.m4a",
]
assert parse_crate(crate_path) == list(crate.references)
def test_parse_crate_ignores_record_without_stop_marker(tmp_path):
crate_path = Path(tmp_path) / "Incomplete.crate"
crate_path.write_bytes(
"Users/sample-user/OneDrive/Jukebox/House/Incomplete.mp3".encode(
"utf-16-le"
)
)
assert parse_crate(crate_path) == []
def test_parse_crate_uses_configured_reference_root(tmp_path):
crate_path = tmp_path / "Migrated.crate"
crate_path.write_bytes(
"Archive/Old Library/House/Track.mp3otrk".encode("utf-16-le")
)
references = parse_crate(crate_path, [Path("/Archive/Old Library")])
assert references[0].path == Path("/Archive/Old Library/House/Track.mp3")
def test_default_path_markers_do_not_contain_a_username():
assert path_markers([]) == ("Users/", "Volumes/")
def test_parse_crates_reuses_configured_roots_for_every_crate(tmp_path):
for name in ("First", "Second"):
(tmp_path / f"{name}.crate").write_bytes(
f"Archive/Jukebox/{name}.mp3otrk".encode("utf-16-le")
)
references = parse_crates(
tmp_path, (root for root in [Path("/Archive/Jukebox")])
)
assert {reference.filename for reference in references} == {
"First.mp3",
"Second.mp3",
}
@pytest.mark.parametrize(
("relative_path", "expected"),
[
("_Serato_/Subcrates/House.crate", CrateKind.STATIC),
("_Serato_/SmartCrates/Warmup.scrate", CrateKind.SMART),
("_Serato_/Subcrates/Compatible by key.crate", CrateKind.SMART),
("fixtures/Unknown.crate", CrateKind.UNKNOWN),
],
)
def test_classify_crate_uses_provenance_and_known_dynamic_name(
tmp_path, relative_path, expected
):
assert classify_crate(tmp_path / relative_path) is expected
def test_smart_crate_preserves_encoded_hierarchy(tmp_path):
crate_path = (
tmp_path / "_Serato_" / "SmartCrates" / "Compatible by key≫≫10A.scrate"
)
crate_path.parent.mkdir(parents=True)
crate_path.write_bytes(b"")
crate = load_crate(crate_path)
assert crate.kind is CrateKind.SMART
assert crate.hierarchy == ("Compatible by key", "10A")
assert crate.display_name == "10A"
assert crate.is_smart_definition
def test_load_library_crates_discovers_scrate_definitions(tmp_path):
smart_folder = tmp_path / "SmartCrates"
static_folder = tmp_path / "Subcrates"
smart_folder.mkdir()
static_folder.mkdir()
(smart_folder / "New EDM.scrate").write_bytes(b"")
(smart_folder / "Re-Drums.scrate").write_bytes(b"")
(static_folder / "House.crate").write_bytes(b"")
crates = load_library_crates(tmp_path)
assert [(crate.path.name, crate.kind) for crate in crates] == [
("New EDM.scrate", CrateKind.SMART),
("Re-Drums.scrate", CrateKind.SMART),
("House.crate", CrateKind.STATIC),
]
+48
View File
@@ -0,0 +1,48 @@
from pathlib import Path
from serato_doctor.database_parser import iter_records, parse_database
def record(tag, payload):
return tag + len(payload).to_bytes(4, "big") + payload
def text_record(tag, value):
return record(tag, value.encode("utf-16-be"))
def test_parse_database_reads_track_paths_and_metadata(tmp_path):
track = b"".join(
[
text_record(b"pfil", "Users/sample/Music/Track.mp3"),
text_record(b"tsng", "Track title"),
text_record(b"tart", "Test artist"),
text_record(b"talb", "Test album"),
text_record(b"tgen", "House"),
]
)
database_path = tmp_path / "database V2"
database_path.write_bytes(
text_record(b"vrsn", "2.0/Test Database") + record(b"otrk", track)
)
database = parse_database(database_path)
assert database.version == "2.0/Test Database"
assert len(database.tracks) == 1
parsed = database.tracks[0]
assert parsed.path == Path("/Users/sample/Music/Track.mp3")
assert parsed.filename == "Track.mp3"
assert parsed.title == "Track title"
assert parsed.artist == "Test artist"
assert parsed.album == "Test album"
assert parsed.genre == "House"
def test_iter_records_ignores_incomplete_trailing_record():
complete = record(b"vrsn", "2.0".encode("utf-16-be"))
incomplete = b"otrk\x00\x00\x00\x10short"
records = list(iter_records(complete + incomplete))
assert records == [(b"vrsn", "2.0".encode("utf-16-be"))]
+57
View File
@@ -0,0 +1,57 @@
from pathlib import Path
from serato_doctor.duplicates import find_duplicate_groups
from serato_doctor.models.duplicate import DuplicateKind
from serato_doctor.models.track import DiskTrack
def track(path):
path = Path(path)
return DiskTrack(path, path.name, 100, path.suffix.lower())
def test_exact_names_in_different_folders_form_a_group():
groups = find_duplicate_groups(
[track("/A/Song.mp3"), track("/B/Song.mp3")]
)
assert len(groups) == 1
assert groups[0].kind is DuplicateKind.EXACT_NAME
assert groups[0].display_name == "Song.mp3"
assert groups[0].extra_files == 1
def test_case_only_difference_is_an_exact_name_duplicate():
groups = find_duplicate_groups(
[track("/A/SONG.MP3"), track("/B/song.mp3")]
)
assert groups[0].kind is DuplicateKind.EXACT_NAME
def test_numeric_suffixes_form_a_suspected_cloud_conflict_group():
groups = find_duplicate_groups(
[
track("/A/Track.mp3"),
track("/B/Track 2.mp3"),
track("/C/Track 3.mp3"),
]
)
assert len(groups) == 1
assert groups[0].kind is DuplicateKind.CLOUD_CONFLICT
assert groups[0].display_name == "Track.mp3"
assert groups[0].extra_files == 2
assert [item.path for item in groups[0].tracks] == [
Path("/A/Track.mp3"),
Path("/B/Track 2.mp3"),
Path("/C/Track 3.mp3"),
]
def test_unrelated_and_single_files_are_not_reported():
groups = find_duplicate_groups(
[track("/A/First.mp3"), track("/B/Second.mp3")]
)
assert groups == ()
+25
View File
@@ -0,0 +1,25 @@
from serato_doctor.hashing import compare_audio_files, sha256_file
def test_equal_files_have_the_same_hash(tmp_path):
first = tmp_path / "First.mp3"
second = tmp_path / "Second.mp3"
first.write_bytes(b"same audio bytes")
second.write_bytes(b"same audio bytes")
result = compare_audio_files((first, second))
assert result["status"] == "identical"
assert sha256_file(first) == sha256_file(second)
def test_different_files_are_not_reported_as_identical(tmp_path):
first = tmp_path / "First.mp3"
second = tmp_path / "Second.mp3"
first.write_bytes(b"version one")
second.write_bytes(b"version two")
result = compare_audio_files((first, second))
assert result["status"] == "different"
assert result["files"][0]["sha256"] != result["files"][1]["sha256"]
+123
View File
@@ -0,0 +1,123 @@
from pathlib import Path
from serato_doctor.health import analyze_health
from serato_doctor.models.crate import Crate, CrateKind
from serato_doctor.models.filesystem import BrokenSymlink
from serato_doctor.models.library import Library
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def reference(filename):
return TrackReference(
Path("Test.crate"), Path("/old/House") / filename, filename
)
def track(filename, folder="House"):
path = Path("/new") / folder / filename
return DiskTrack(path, filename, 100, path.suffix.lower())
def test_health_report_exposes_each_metric():
library = Library.build(
[
reference("Found.mp3"),
reference("Conflict.mp3"),
reference("Absent.mp3"),
],
[
track("Found.mp3"),
track("Conflict 2.mp3"),
track("Conflict 2.mp3", "Backup"),
track("Unused.mp3"),
],
)
report = analyze_health(library)
assert report.score == 33.3
assert report.score_basis == (
"Resolved non-dynamic references / non-dynamic references scored"
)
assert report.total_references == 3
assert report.scored_references == 3
assert report.healthy_references == 1
assert report.missing_references == 2
assert report.unique_missing_filenames == 2
assert report.disk_tracks == 4
assert report.duplicate_filename_groups == 1
assert report.duplicate_files == 1
assert report.unused_tracks == 3
assert report.suggested_matches == 1
def test_empty_library_has_no_health_score():
report = analyze_health(Library.build([], []))
assert report.score is None
assert report.total_references == 0
assert report.scored_references == 0
assert not report.database_present
assert report.tracks_missing_from_database == 0
def test_smart_crate_references_are_reported_but_not_scored():
static_reference = reference("Found.mp3")
smart_reference = TrackReference(
Path("SmartCrates/Dynamic.scrate"),
Path("/old/House/Dynamic.mp3"),
"Dynamic.mp3",
)
crates = [
Crate(Path("Subcrates/Static.crate"), (static_reference,), CrateKind.STATIC),
Crate(
Path("SmartCrates/Dynamic.scrate"),
(smart_reference,),
CrateKind.SMART,
),
]
library = Library.from_crates(crates, [track("Found.mp3")])
report = analyze_health(library)
assert report.score == 100.0
assert report.total_references == 2
assert report.scored_references == 1
assert report.missing_references == 0
assert report.static_crates == 1
assert report.smart_crates == 1
assert report.smart_crate_containers == 0
assert report.dynamic_references_excluded == 1
def test_health_reports_cloud_conflicts_separately_from_exact_duplicates():
library = Library.build(
[],
[
track("Track.mp3", "Original"),
track("Track 2.mp3", "Conflict"),
track("Copy.mp3", "First"),
track("Copy.mp3", "Second"),
],
)
report = analyze_health(library)
assert report.duplicate_filename_groups == 1
assert report.duplicate_files == 1
assert report.suspected_cloud_conflict_groups == 1
assert report.suspected_cloud_conflict_files == 1
def test_health_reports_broken_symlinks_without_changing_score():
library = Library.build(
[reference("Found.mp3")],
[track("Found.mp3")],
[BrokenSymlink(Path("/music/Broken.mp3"), Path("missing.mp3"))],
)
report = analyze_health(library)
assert report.score == 100.0
assert report.broken_symlinks == 1
+24
View File
@@ -0,0 +1,24 @@
from pathlib import Path
from serato_doctor.models.library import Library
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def test_library_reconciles_references_by_filename():
crate = Path("House.crate")
references = [
TrackReference(crate, Path("/old/Found.mp3"), "Found.mp3"),
TrackReference(crate, Path("/old/Missing.mp3"), "Missing.mp3"),
]
tracks = [DiskTrack(Path("/new/Found.mp3"), "Found.mp3", 10, ".mp3")]
results = Library.build(references, tracks).reconcile_by_filename()
assert [result.exists_by_filename for result in results] == [True, False]
assert results[1].as_row() == {
"crate": "House.crate",
"serato_path": "/old/Missing.mp3",
"filename": "Missing.mp3",
"exists_by_filename": False,
}
+34
View File
@@ -0,0 +1,34 @@
import logging
from serato_doctor.logging import LOGGER_NAME, configure_logging
def test_logging_is_quiet_by_default(capsys):
logger = configure_logging()
logger.info("not visible")
assert capsys.readouterr().err == ""
def test_logging_writes_aggregate_progress_to_file(tmp_path):
log_path = tmp_path / "scan.log"
logger = configure_logging(log_file=log_path)
logger.info("Scanned %d disk tracks", 10)
contents = log_path.read_text(encoding="utf-8")
assert "INFO Scanned 10 disk tracks" in contents
def test_reconfiguring_logging_replaces_handlers():
configure_logging(verbose=True)
logger = configure_logging(verbose=True)
active_handlers = [
handler
for handler in logger.handlers
if not isinstance(handler, logging.NullHandler)
]
assert logger.name == LOGGER_NAME
assert len(active_handlers) == 1
+62
View File
@@ -0,0 +1,62 @@
from pathlib import Path
from serato_doctor.matching import MatchingEngine, score_candidate
from serato_doctor.models.reference import TrackReference
from serato_doctor.models.track import DiskTrack
def reference(filename, folder="House"):
return TrackReference(
Path("Test.crate"), Path("/old") / folder / filename, filename
)
def track(filename, folder="House"):
path = Path("/new") / folder / filename
return DiskTrack(path, filename, 100, path.suffix.lower())
def test_exact_candidate_has_full_evidence_score():
match = score_candidate(reference("Track.mp3"), track("Track.mp3"))
assert match.score == 90
assert match.max_score == 90
assert match.score_percent == 100.0
assert all(item.matched for item in match.evidence)
def test_cloud_conflict_suffix_is_explained():
match = score_candidate(reference("Track.mp3"), track("Track 2.mp3"))
assert match.score == 80
assert match.score_percent == 88.9
assert match.evidence[0].explanation == (
"Filename matches after removing a numeric conflict suffix"
)
def test_case_normalized_match_scores_below_exact():
match = score_candidate(reference("TRACK.MP3"), track("track.mp3"))
assert match.score == 85
assert match.evidence[0].points == 55
def test_engine_omits_unrelated_filenames():
engine = MatchingEngine([track("Different.mp3")])
assert engine.candidates_for(reference("Missing.mp3")) == ()
def test_ambiguous_candidates_are_ranked_deterministically():
engine = MatchingEngine(
[track("Track 2.mp3", "Other"), track("Track 3.mp3", "House")]
)
matches = engine.candidates_for(reference("Track.mp3"))
assert [match.track.filename for match in matches] == [
"Track 3.mp3",
"Track 2.mp3",
]
assert [match.score for match in matches] == [80, 60]
+101
View File
@@ -0,0 +1,101 @@
from pathlib import Path
import pytest
from serato_doctor.repair import (
BACKUP_FOLDER,
apply_duplicate_repair,
apply_duplicate_repair_batch,
list_backups,
plan_duplicate_repair,
restore_backup,
rotate_backups,
)
def library(tmp_path):
serato = tmp_path / "_Serato_"
crate = serato / "Subcrates" / "House.crate"
crate.parent.mkdir(parents=True)
crate.write_bytes(b"crate-data")
(serato / "database V2").write_bytes(b"database-data")
music = tmp_path / "Music"
keeper = music / "Main" / "Track.mp3"
duplicate = music / "Old" / "Track.mp3"
keeper.parent.mkdir(parents=True)
duplicate.parent.mkdir(parents=True)
keeper.write_bytes(b"keeper")
duplicate.write_bytes(b"duplicate")
return serato, keeper, duplicate
def test_duplicate_repair_backs_up_then_preserves_old_path_as_link(tmp_path):
serato, keeper, duplicate = library(tmp_path)
plan = plan_duplicate_repair(keeper, (keeper, duplicate), serato)
receipt = apply_duplicate_repair(plan, serato, backup_limit=3)
assert duplicate.is_symlink()
assert duplicate.resolve() == keeper.resolve()
assert (receipt.backup / "manifest.json").is_file()
assert any((receipt.backup / "serato-metadata").rglob("House.crate"))
assert any((receipt.backup / "serato-metadata").rglob("database V2"))
history = list_backups(serato)
assert len(history) == 1
assert history[0].status == "ready"
assert history[0].size_bytes > 0
restored = restore_backup(receipt.backup)
assert restored == (duplicate.resolve(),)
assert not duplicate.is_symlink()
assert duplicate.read_bytes() == b"duplicate"
assert list_backups(serato)[0].status == "restored"
def test_plan_rejects_keeper_outside_duplicate_group(tmp_path):
serato, keeper, duplicate = library(tmp_path)
outsider = tmp_path / "outsider.mp3"
outsider.write_bytes(b"other")
with pytest.raises(ValueError, match="must belong"):
plan_duplicate_repair(outsider, (keeper, duplicate), serato)
def test_backup_rotation_can_be_limited_or_unlimited(tmp_path):
root = tmp_path / BACKUP_FOLDER
for name in ("001", "002", "003"):
backup = root / name
backup.mkdir(parents=True)
(backup / "manifest.json").write_text("{}", encoding="utf-8")
rotate_backups(root, None)
assert len(list(root.iterdir())) == 3
rotate_backups(root, 2)
assert {path.name for path in root.iterdir()} == {"002", "003"}
def test_batch_repair_uses_one_backup_and_restores_every_group(tmp_path):
serato, first_keeper, first_duplicate = library(tmp_path)
second_keeper = tmp_path / "Music" / "Main" / "Other.mp3"
second_duplicate = tmp_path / "Music" / "Old" / "Other.mp3"
second_keeper.write_bytes(b"other-keeper")
second_duplicate.write_bytes(b"other-duplicate")
plans = (
plan_duplicate_repair(
first_keeper, (first_keeper, first_duplicate), serato
),
plan_duplicate_repair(
second_keeper, (second_keeper, second_duplicate), serato
),
)
receipt = apply_duplicate_repair_batch(plans, serato, backup_limit=10)
assert first_duplicate.is_symlink()
assert second_duplicate.is_symlink()
assert len(list((serato / BACKUP_FOLDER).iterdir())) == 1
assert len(receipt.replaced) == 2
restore_backup(receipt.backup)
assert first_duplicate.read_bytes() == b"duplicate"
assert second_duplicate.read_bytes() == b"other-duplicate"
+34
View File
@@ -0,0 +1,34 @@
import runpy
from pathlib import Path
from serato_doctor.crate_parser import load_library_crates
from serato_doctor.models.crate import CrateKind
from serato_doctor.models.library import Library
from serato_doctor.scanner import scan_audio
def test_generated_sample_library_has_expected_scenario(tmp_path):
generator_path = (
Path(__file__).parents[1] / "samples" / "small-library" / "generate.py"
)
build_sample = runpy.run_path(str(generator_path))["build_sample"]
sample_root = build_sample(tmp_path / "sample")
crates = load_library_crates(sample_root / "Serato" / "_Serato_")
tracks = scan_audio(sample_root / "Music")
library = Library.from_crates(crates, tracks)
results = library.reconcile_by_filename()
missing = {
result.reference.filename
for result in results
if not result.exists_by_filename
}
crate_files = list((sample_root / "Serato").rglob("*.crate"))
smart_files = list((sample_root / "Serato").rglob("*.scrate"))
assert len(crate_files) + len(smart_files) == 7
assert len(library.references) == 13
assert len(tracks) == 10
assert missing == {"Missing.mp3", "Old Name.mp3"}
assert sum(crate.kind is CrateKind.STATIC for crate in crates) == 5
assert sum(crate.kind is CrateKind.SMART for crate in crates) == 2
+50
View File
@@ -0,0 +1,50 @@
from pathlib import Path
from serato_doctor.scanner import scan_audio, scan_filesystem
def test_scan_audio_finds_supported_files(tmp_path):
music = tmp_path / "music"
nested = music / "House"
nested.mkdir(parents=True)
(nested / "First.MP3").write_bytes(b"synthetic audio")
(nested / "Second.flac").write_bytes(b"fixture")
(nested / "notes.txt").write_text("not audio", encoding="utf-8")
tracks = scan_audio(music)
assert {track.filename for track in tracks} == {"First.MP3", "Second.flac"}
first = next(track for track in tracks if track.filename == "First.MP3")
assert first.suffix == ".mp3"
assert first.size == len(b"synthetic audio")
def test_scan_filesystem_reports_broken_symlink(tmp_path):
music = tmp_path / "music"
music.mkdir()
link = music / "Missing.mp3"
link.symlink_to("not-there.mp3")
result = scan_filesystem(music)
assert result.tracks == ()
assert len(result.broken_symlinks) == 1
assert result.broken_symlinks[0].path == link
assert result.broken_symlinks[0].target == Path("not-there.mp3")
def test_valid_audio_symlink_is_scanned_normally(tmp_path):
music = tmp_path / "music"
music.mkdir()
target = music / "Target.mp3"
target.write_bytes(b"synthetic audio")
link = music / "Linked.mp3"
link.symlink_to(target)
result = scan_filesystem(music)
assert {track.filename for track in result.tracks} == {
"Linked.mp3",
"Target.mp3",
}
assert result.broken_symlinks == ()
+148
View File
@@ -0,0 +1,148 @@
import runpy
from pathlib import Path
import pytest
from serato_doctor.web import (
STATIC_FILES,
_file_token,
_verified_audio,
analyze_paths,
duplicate_repair,
duplicate_repair_batch,
)
def test_web_analysis_uses_production_health_pipeline(tmp_path):
generator = runpy.run_path(
str(Path(__file__).parents[1] / "samples/small-library/generate.py")
)
sample = generator["build_sample"](tmp_path / "sample")
result = analyze_paths(
sample / "Serato" / "_Serato_", sample / "Music"
)
assert result["score"] == 77.8
assert result["disk_tracks"] == 10
assert result["missing_references"] == 2
assert result["static_crates"] == 5
assert result["smart_crates"] == 2
assert result["database_entries"] == 10
assert result["database_library_matches"] == 10
assert result["database_missing_paths"] == 0
assert result["database_missing_unique_filenames"] == 0
assert result["tracks_missing_from_database"] == 0
assert result["details"]["old_crate_references"]["total"] == 2
assert result["details"]["old_crate_references"]["items"][0]["crate"]
assert result["details"]["suggested_matches"]["total"] == 0
assert result["details"]["unused_tracks"]["total"] == 1
def test_web_analysis_rejects_missing_folders(tmp_path):
with pytest.raises(ValueError, match="Serato folder does not exist"):
analyze_paths(tmp_path / "missing", tmp_path)
def test_web_static_assets_are_declared_and_packaged():
asset_root = Path(__file__).parents[1] / "serato_doctor" / "webui"
assert set(STATIC_FILES) == {
"/",
"/app.css",
"/recovery.css",
"/layout-fixes.css",
"/app.js",
}
assert all((asset_root / filename).is_file() for filename, _ in STATIC_FILES.values())
html = (asset_root / "index.html").read_text(encoding="utf-8")
assert "Missing tracks in Serato" in html
assert "Old crate references" in html
assert "Choose a diagnostic" in html
assert "Backup recovery" in html
assert "Heres exactly what will happen" in html
assert 'id="batch-preview-modal"' in html
assert 'data-detail="database_missing_tracks"' in html
assert 'data-detail="old_crate_references"' in html
assert html.count('class="info-button"') >= 10
def test_duplicate_repair_preview_does_not_change_files(tmp_path):
serato = tmp_path / "_Serato_"
serato.mkdir()
music = tmp_path / "Music"
first = music / "A" / "Track.mp3"
second = music / "B" / "Track.mp3"
first.parent.mkdir(parents=True)
second.parent.mkdir(parents=True)
first.write_bytes(b"first")
second.write_bytes(b"second")
result = duplicate_repair(
serato, music, first, (first, second), backup_limit=10
)
assert result["applied"] is False
assert result["database_v2_modified"] is False
assert second.read_bytes() == b"second"
assert not second.is_symlink()
def test_audio_preview_tokens_only_open_signed_audio(tmp_path):
track = tmp_path / "Track.mp3"
track.write_bytes(b"audio")
token = _file_token(track)
assert _verified_audio(token) == track
with pytest.raises(ValueError, match="Invalid or expired"):
_verified_audio(token + "changed")
def test_duplicate_details_include_preview_and_finder_controls(tmp_path):
serato = tmp_path / "_Serato_"
serato.mkdir()
music = tmp_path / "Music"
first = music / "A" / "Track.mp3"
second = music / "B" / "Track.mp3"
first.parent.mkdir(parents=True)
second.parent.mkdir(parents=True)
first.write_bytes(b"first")
second.write_bytes(b"second")
result = analyze_paths(serato, music)
group = result["details"]["duplicate_filenames"]["items"][0]
assert len(group["file_previews"]) == 2
assert group["file_previews"][0]["audio_url"].startswith("/api/audio?")
assert group["file_previews"][0]["reveal_token"]
def test_batch_preview_combines_approved_groups_without_changes(tmp_path):
serato = tmp_path / "_Serato_"
serato.mkdir()
music = tmp_path / "Music"
choices = []
for filename in ("First.mp3", "Second.mp3"):
keeper = music / "A" / filename
duplicate = music / "B" / filename
keeper.parent.mkdir(parents=True, exist_ok=True)
duplicate.parent.mkdir(parents=True, exist_ok=True)
keeper.write_bytes(b"keeper")
duplicate.write_bytes(b"duplicate")
choices.append(
{"keeper": str(keeper), "group_files": [str(keeper), str(duplicate)]}
)
result = duplicate_repair_batch(
serato, music, choices, backup_limit=10
)
assert result["applied"] is False
assert result["choice_count"] == 2
assert len(result["replaced"]) == 2
assert len(result["decisions"]) == 2
assert result["decisions"][0]["keeper"].endswith("First.mp3")
assert result["decisions"][0]["hash_status"] == "different"
assert len(result["decisions"][0]["replaced"]) == 1
assert all(not Path(choice["group_files"][1]).is_symlink() for choice in choices)