coolpoodle commited on
Commit
90884df
·
verified ·
1 Parent(s): d614f1c

code and training scripts

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +5 -35
  2. .gitignore +43 -0
  3. DATA.md +60 -0
  4. FINDINGS.md +245 -0
  5. LICENSE +201 -0
  6. MODEL_CARD.md +59 -0
  7. NOTICE +26 -0
  8. README.md +145 -0
  9. REPORT_AUDIO_PREPEND.md +14 -0
  10. REPORT_REFERENCE_RANKED_POOL.md +30 -0
  11. REPORT_REFERENCE_STYLE.md +32 -0
  12. REPRODUCING.md +90 -0
  13. capture_generated_tokens.py +460 -0
  14. checkpoints/README.md +35 -0
  15. configs/acoustic-fim-v2.yaml +41 -0
  16. configs/audio-continuation-v1.yaml +25 -0
  17. configs/audio-fim-v1.yaml +18 -0
  18. configs/audio-prepend-v1.yaml +18 -0
  19. configs/autonomous-interim678-v1.yaml +20 -0
  20. configs/baseline.yaml +52 -0
  21. configs/captured-state-style-continuation-v1.yaml +36 -0
  22. configs/continuation-tier-a.yaml +25 -0
  23. configs/external-flow-encoder-interim-678-v1.yaml +63 -0
  24. configs/external-flow-encoder-v1.yaml +56 -0
  25. configs/flow-encoder-v1.yaml +78 -0
  26. configs/flow-inpaint-v1.yaml +67 -0
  27. configs/flow-prepend-v1.yaml +66 -0
  28. configs/inversion-e3.yaml +139 -0
  29. configs/inversion-v1.yaml +94 -0
  30. configs/inversion-v2.yaml +175 -0
  31. configs/learned-audio-continuation-v1.yaml +77 -0
  32. configs/long-reference-style-v1.yaml +26 -0
  33. configs/native-state-distill.yaml +46 -0
  34. configs/native-state-multiclip.yaml +44 -0
  35. configs/native-state-stage0.yaml +39 -0
  36. configs/native-state-stage1-local.yaml +32 -0
  37. configs/native-state-stage1.yaml +40 -0
  38. configs/native-token-adapter-v1.yaml +33 -0
  39. configs/objective-only-v1.yaml +170 -0
  40. configs/prefix-seed-search-v1.yaml +56 -0
  41. configs/prompt-free-v1.yaml +19 -0
  42. configs/promptfree-bestofn-v1.yaml +29 -0
  43. configs/reference-guided-append-v1.yaml +14 -0
  44. configs/reference-ranked-pool-v1.yaml +27 -0
  45. configs/reference-style-v1.yaml +20 -0
  46. configs/sample-bridge-v1.yaml +23 -0
  47. configs/waveform-causal-continuation-v1.yaml +44 -0
  48. configs/waveform-right-context-prepend-v1.yaml +39 -0
  49. data/laion_disco/UPSTREAM_README.md +54 -0
  50. data/laion_disco/candidates.summary.json +28 -0
.gitattributes CHANGED
@@ -1,35 +1,5 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
1
+ * text=auto eol=lf
2
+ *.wav binary
3
+ *.safetensors binary
4
+ *.png binary
5
+ *.jpg binary
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
.gitignore ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # --- Python ---
2
+ .venv/
3
+ venv/
4
+ __pycache__/
5
+ *.py[cod]
6
+ .pytest_cache/
7
+ *.egg-info/
8
+ build/
9
+ dist/
10
+
11
+ # --- models / weights / audio (never commit) ---
12
+ models/
13
+ artifacts/
14
+ checkpoints/**/*.safetensors
15
+ *.safetensors
16
+ *.pt
17
+ *.pth
18
+ *.ckpt
19
+ *.bin
20
+ *.wav
21
+ *.mp3
22
+ *.m4a
23
+ *.flac
24
+ *.npy
25
+ *.npz
26
+
27
+ # --- downloaded dataset audio (keep metadata, never the audio) ---
28
+ data/**/files/
29
+ data/**/canonical/
30
+ data/**/*.wav
31
+
32
+ # --- secrets / credentials (never commit) ---
33
+ .env
34
+ *.pem
35
+ *.key
36
+ id_rsa*
37
+ id_ed25519*
38
+ *hf_token*
39
+ **/known_user22_* # private author-song exclusion hashes/filenames
40
+
41
+ # --- os cruft ---
42
+ .DS_Store
43
+ Thumbs.db
DATA.md ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Data
2
+
3
+ Music3Lab ships **dataset metadata only**. No audio of any kind is included or
4
+ redistributed.
5
+
6
+ ## LAION-DISCO-12M (external real-music corpus)
7
+
8
+ [LAION-DISCO-12M](https://huggingface.co/datasets/laion/LAION-DISCO-12M) is an
9
+ **index** dataset: it contains YouTube video IDs and tags, licensed **Apache-2.0**
10
+ for the metadata, and explicitly contains **no audio samples**. Audio referenced
11
+ by those IDs is *not* part of the dataset and is *not* redistributed here.
12
+
13
+ ### What is shipped in this repo (`data/laion_disco/`)
14
+
15
+ - `interim_tranche_678/manifest.jsonl` — the 678 accepted tracks as records of:
16
+ YouTube `source_id`/`source_url`, LAION `source_dataset_revision`, canonical
17
+ audio content hashes (`canonical_sha256`, `pcm_sha256`), sample rate / channels
18
+ / frame counts, and split assignment. **No audio.**
19
+ - `interim_tranche_678/splits.json` — deterministic, source-level train/val/heldout
20
+ split (algorithm, seed, assignment, counts).
21
+ - `interim_tranche_678/summary.json`, `candidates.summary.json`,
22
+ `tranche.stopped.json`, `environment.json` — provenance and stop-reason records.
23
+ - `UPSTREAM_README.md` — the corpus builder's own notes.
24
+
25
+ ### What is NOT shipped
26
+
27
+ - ❌ Any downloaded audio (`files/*.wav`, `canonical/`).
28
+ - ❌ The full 20k candidate list is included only as a summary; the raw
29
+ `candidates.jsonl` and the private author-song exclusion hashes
30
+ (`known_user22_*`) are **not** published.
31
+
32
+ ### Reproducing the corpus
33
+
34
+ `scripts/data/laion_ingest.py` deterministically selects candidates from the
35
+ pinned LAION revision (`6e7bf3758a77301e46a715af894fefd79bb1da53`), then uses
36
+ `yt-dlp` + `ffmpeg` to fetch and canonicalize audio **on your own machine**.
37
+ `scripts/data/laion_freeze_interim.py` freezes an immutable, hash-verified
38
+ tranche and split. Both scripts contain instance-specific absolute paths from the
39
+ original run and need their `ROOT`/paths adjusted before use.
40
+
41
+ > ⚠️ **You are responsible** for complying with YouTube's Terms of Service and
42
+ > applicable copyright when downloading any audio. Music3Lab neither hosts nor
43
+ > distributes that audio. The original run stopped at 678/2,000 tracks when
44
+ > YouTube returned bot-verification challenges — expect the same and throttle
45
+ > accordingly.
46
+
47
+ ## The author's own songs (private)
48
+
49
+ The project also evaluated on the author's own tracks (8, then 22 after
50
+ deduplication). These are **private by default** and are **not** in this repo,
51
+ not in the LAION splits, and were **never** used for training or checkpoint
52
+ selection — only as out-of-distribution evaluation. Only their content **hashes**
53
+ were used, to guarantee they never leaked into training. If you want a public
54
+ demo, publish only short excerpts of audio you own and are licensed to share.
55
+
56
+ ## MiniMax-Music3 teacher data
57
+
58
+ Some experiments use WAV/latent/token pairs *captured from* MiniMax-Music3
59
+ generations. These captures are derivatives of the base model and are **not**
60
+ shipped. `scripts/` can regenerate them from the model you download yourself.
FINDINGS.md ADDED
@@ -0,0 +1,245 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # MiniMax Music 3 `dav.pth` investigation
2
+
3
+ ## Conclusion
4
+
5
+ The released `dav.pth` does **not** contain the native Music 3 waveform-to-RVQ tokenizer. It contains a continuous
6
+ Flow-VAE analysis/synthesis checkpoint: a waveform convolutional `encoder`, Gaussian posterior heads (`mean_proj` and
7
+ `logs_proj`), continuous `flow` transforms, and a waveform `decoder`. There is no RVQ/VQ quantizer and there are no
8
+ acoustic codebook embeddings. The `encoder.*` name is therefore real but is not the discrete encoder required by the
9
+ proposed continuation path.
10
+
11
+ The safe CPU-only inspection of release revision `fbdf52fbaaca799592917417eb05f1899f1255ec` found:
12
+
13
+ | Prefix | Tensor count | Representative evidence |
14
+ |---|---:|---|
15
+ | `encoder.*` | 119 | `encoder.block.0.weight_v [64,1,7]`; final trunk `encoder.block.6.weight_v [1024,1024,3]` |
16
+ | `mean_proj.*` | 2 | `mean_proj.weight [64,1024,1]` |
17
+ | `logs_proj.*` | 2 | `logs_proj.weight [64,1024,1]` |
18
+ | `dec_in_proj.*` | 2 | `dec_in_proj.weight [1024,64,1]` |
19
+ | `decoder.*` | 119 | first convolution `[1536,1024,7]`; final convolution `[1,96,7]` |
20
+ | `flow.*` | 304 | `flow.flows.0.pre.weight [256,32,1]` through `flow.flows.6.post.weight [32,256,1]` |
21
+
22
+ That is 548 tensors, all FP32, containing 122,904,034 values (491,616,136 tensor bytes). The file is 491,817,450
23
+ bytes and has SHA-256 `52adde6c6c52cca872f549449cd7b608677d1e69f4a98eee22da3403e63b56e8`. It is a flat
24
+ `OrderedDict`: there is no `generator` key, no `62000_generator` key or wrapper, and zero names matching
25
+ `quantizer`, `vq`, `rvq`, `codebook`, `codec`, or `tokenizer`. It also contains no serialized config or architecture
26
+ metadata. Tensor shapes plus released source are sufficient to instantiate the decoder, but cannot instantiate a
27
+ quantizer that has neither parameters nor configuration.
28
+
29
+ The reported community `62000_generator` wrapper is not present in this exact released file. `inspect_dav.py` still
30
+ recursively supports `generator`, `62000_generator`, and `state_dict`-style wrappers so a different checkpoint can be
31
+ checked without assuming it has the release layout. Names, shapes, and generic config can flag a candidate weight set,
32
+ but the inspector deliberately keeps `native_discrete_tokenizer_complete` and `can_encode_native_music3_tokens` false
33
+ until an exact compatible executable architecture/API has been implemented and verified.
34
+
35
+ ## The actual token and rendering paths
36
+
37
+ Music 3 uses eight discrete streams at 25 frames/s. They are language-model symbols used to produce continuous
38
+ conditioning hidden states; they are **not** DAV latent codes and are never passed directly to the DAV decoder.
39
+
40
+ ```text
41
+ caption + lyrics + <|audio_start|>
42
+ |
43
+ v
44
+ Global Qwen LM: c0 in [0, 16383] (vocabulary token c0 + 151675)
45
+ |
46
+ v
47
+ Local RVQ-depth LM: c1..c7, each in [0, 1023]
48
+ |
49
+ +--> frame token row [c0,c1,...,c7]
50
+ |
51
+ +--> global hidden + 7 depth hiddens = 8 * 4096
52
+ frame_hiddens [1, frames, 32768]
53
+ |
54
+ v
55
+ condition projection + flow-matching DiT
56
+ |
57
+ v
58
+ continuous Flow-VAE latent [B,128,T]
59
+ |
60
+ v
61
+ DAV dec_in_proj + decoder -> waveform
62
+ ```
63
+
64
+ The token trajectory convention used by the provided comparator is `[frames, 8]`; `[8, frames]` is accepted only
65
+ when explicitly declared as `codebooks_first`. No heuristic transposition is performed.
66
+
67
+ The per-frame autoregressive details are:
68
+
69
+ 1. The global LM samples only `<|audio_end|>` or the 16,384 c0 vocabulary IDs beginning at offset 151675.
70
+ 2. The local depth model begins with the projected global hidden and the projected c0 embedding, then samples c1
71
+ through c7 autoregressively. It has seven 1,024-way output heads.
72
+ 3. For feedback to the global LM, c0 uses the global token embedding at `c0 + 151675`. Each residual code uses the
73
+ Qwen checkpoint's `model.audio_extra_embedding.weight` in seven disjoint 1,024-row bands. Those seven embeddings
74
+ are summed with c0 and scaled by `1/sqrt(8)`. This is an **LM embedding table**, not the absent waveform
75
+ quantizer's acoustic centroids.
76
+ 4. Codebooks are frame-aligned (one row contains all eight codes); there is no delayed-codebook layout in this path.
77
+ Conditional and classifier-free-unconditional rows are paired during sampling. The first decode immediately after
78
+ `<|audio_start|>` primes the feedback loop and is not emitted as an acoustic hidden frame. Each subsequent emitted
79
+ frame concatenates one 4096-wide global hidden with seven 4096-wide local hiddens.
80
+ 5. The 25 Hz rate is explicit in Diffusers and also follows `24000 / 960` in the condition encoder configuration.
81
+
82
+ Special IDs are fixed by the music tokenizer:
83
+
84
+ | Token | ID |
85
+ |---|---:|
86
+ | `<|im_start|>` | 151644 |
87
+ | `<|im_end|>` | 151645 |
88
+ | `<|audio_cfg|>` | 151654 |
89
+ | `<|audio_start|>` | 151669 |
90
+ | `<|audio_end|>` | 151670 |
91
+ | `<|caption_start|>` | 151671 |
92
+ | `<|caption_end|>` | 151672 |
93
+ | `<|lyrics_start|>` | 151673 |
94
+ | `<|lyrics_end|>` | 151674 |
95
+ | first c0/audio-code vocabulary ID | 151675 |
96
+
97
+ The Qwen shards contain the global audio-symbol embeddings plus the local autoregressive depth decoder
98
+ (`model.audio_extra_embedding.*` and `model.audio_decoder.*`). They contain no waveform encoder, nearest-neighbor
99
+ quantizer, RVQ codebooks, or equivalent audio-tokenizer module. A model that can *predict* token IDs from text is not
100
+ therefore able to infer those IDs from a waveform.
101
+
102
+ ## Successful official H100 generation capture
103
+
104
+ A pinned end-to-end Diffusers run succeeded on the remote NVIDIA H100 using model revision
105
+ `fbdf52fbaaca799592917417eb05f1899f1255ec`, Diffusers revision
106
+ `90b4e34e79a86ec5e7f2437634fe95ecd2108796`, CUDA with bfloat16 weights, and seed 7. The exact request used prompt
107
+ `Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass.`, lyrics
108
+ `[instrumental]`, requested duration 1.0 seconds, and 30 inference steps. It made 26 calls to the official
109
+ `_generate_depth_codes`: one priming call followed by 25 emitted frames. After skipping only that priming row,
110
+ `music3_internal_tokens.npy` is a genuine internal `[25,8]` trajectory at 25 Hz.
111
+
112
+ The observed inclusive per-codebook ranges were:
113
+
114
+ | Codebook | Minimum | Maximum | Legal range |
115
+ |---|---:|---:|---:|
116
+ | c0 | 1012 | 16163 | 0..16383 |
117
+ | c1 | 95 | 984 | 0..1023 |
118
+ | c2 | 25 | 1005 | 0..1023 |
119
+ | c3 | 42 | 950 | 0..1023 |
120
+ | c4 | 83 | 941 | 0..1023 |
121
+ | c5 | 2 | 967 | 0..1023 |
122
+ | c6 | 3 | 984 | 0..1023 |
123
+ | c7 | 67 | 957 | 0..1023 |
124
+
125
+ The corresponding official renderer output, `music3_generated_1s.wav`, is 44.1 kHz stereo with 44,032 samples per
126
+ channel (0.9984580499 seconds). The reusable `capture_generated_tokens.py` utility reproduces this observation by
127
+ temporarily monkeypatching the official depth-code function, validating its paired `[2,8]` result, restoring the
128
+ function after generation, and saving only calls after the priming row. Before model loading it resolves the requested
129
+ Hub revision to a concrete snapshot commit, verifies the exact Diffusers commit, rejects dirty tracked files in a
130
+ source checkout, and hashes both the relevant Diffusers source file and the capture script. It writes temporary WAV,
131
+ NPY, and JSON files, atomically replaces the two data artifacts, and replaces metadata last. The metadata records the
132
+ runtime and request plus SHA-256 digests of the final WAV and NPY, so an interrupted mixed-generation bundle cannot
133
+ validate. It records `"wav_reencoding_performed": false` because this is an internal generation capture, not WAV
134
+ analysis.
135
+
136
+ For the final canonical rerun, the WAV SHA-256 is
137
+ `e52c8884acdfbe9a71badac1ef410a5e5b23512dfc3d56a6f76228adaae87e11`, the NPY SHA-256 is
138
+ `42de27f795e8a1f1b97ae85fd3a540d432a8b178fdc63f7037bf0eea08c61350`, and the metadata JSON SHA-256 is
139
+ `7232d979f21cdf769e96c047b11ee31bbe28a9ee30ec6cd6960455c44b71aa02`. Metadata embeds the first two digests
140
+ along with capture-script SHA-256 `e4b6ecb87652deaa987a722e7c281aec2196295a717066aec05ef195361c0f1f`
141
+ and relevant Diffusers encoder-source SHA-256
142
+ `247ca4492a531c6593485b9fb11503da8607d03315c2f8b313dc6ae44985336b`.
143
+
144
+ This run proves that the documented token shape, ranges, frame rate, priming alignment, and generation/rendering path
145
+ are executable. It does **not** supply the missing WAV-to-tokenizer path. Calling `encode_audio()` for the generated
146
+ WAV is still blocked by the released DAV capabilities, so there is no recovered token trajectory to compare.
147
+ Per-codebook WAV re-encoding agreement percentages therefore remain unavailable; comparing
148
+ `music3_internal_tokens.npy` with itself would only be a self-comparison and must not be reported as re-encoding
149
+ agreement.
150
+
151
+ ## What conversion omits
152
+
153
+ The Diffusers converter maps only `dec_in_proj.*` and `decoder.*` from `dav.pth`. Its explicit conversion dictionary
154
+ and decoder loop leave 427 of the 548 tensors unmapped: all 119 `encoder.*`, both `mean_proj.*`, both `logs_proj.*`,
155
+ and all 304 `flow.*` tensors. SGLang-Omni makes the same decoder-only selection with prefixes
156
+ `("dec_in_proj.", "decoder.")`. That omission is real, but restoring those keys would expose only continuous DAV
157
+ analysis/posterior operations. It would not restore the absent RVQ quantizer/codebooks.
158
+
159
+ ## Answers to the six requested questions
160
+
161
+ 1. **Is the original audio encoder present?** A waveform analysis encoder is present as `encoder.*`, but the native
162
+ discrete/RVQ tokenizer encoder is not. Its outputs feed continuous `mean_proj`/`logs_proj` and `flow` modules.
163
+ 2. **Is the RVQ quantizer present?** No. There are zero matching quantizer/RVQ/VQ parameters and no acoustic
164
+ codebooks in `dav.pth`. The Qwen local “RVQ depth decoder” predicts residual token IDs; it is not a waveform
165
+ quantizer.
166
+ 3. **Can arbitrary WAV audio be converted into valid Music 3 tokens?** No, not with the released weights. Calling
167
+ `encode_audio()` raises `NativeTokenizerUnavailableError` before opening the WAV and never invents token IDs.
168
+ 4. **Do re-encoded tokens match Music 3 internal tokens?** This experiment is blocked because no re-encoded tokens
169
+ can be produced. Reporting per-codebook percentages would be fabricated. `native_token_compatibility.py` performs
170
+ strict range/layout validation and exact per-codebook comparison once both real trajectories exist.
171
+ 5. **Can those tokens seed continuation?** No token trajectory can be recovered, so continuation cannot be seeded.
172
+ The current released inference entry points also start from text plus `<|audio_start|>` rather than accepting an
173
+ existing token/LM-cache history. Even if a tokenizer is released later, continuation must replay each complete
174
+ eight-code frame through the exact feedback embedding/depth-hidden path and preserve the priming alignment.
175
+ 6. **What is missing?** The compatible waveform-to-code encoder/quantizer implementation, its RVQ acoustic codebook
176
+ weights and config, plus an official existing-audio history/prefill interface. Nothing here justifies retraining or
177
+ substituting a generic codec.
178
+
179
+ There is intentionally no `continue_audio.py`: without native tokens, such a CLI could only ignore the input audio,
180
+ mislabel continuous latents as tokens, or use an incompatible replacement codec. All three would violate the stated
181
+ conditioning contract. Likewise, a token round trip is impossible: Music 3 renders `frame_hiddens` through a
182
+ flow-matching model and DAV, not discrete codes directly through DAV.
183
+
184
+ ## Reproducible checks
185
+
186
+ The inspection, comparison, blocked-encode, and test paths are CPU-only. Checkpoint loading uses
187
+ `torch.load(..., weights_only=True, map_location="cpu")`. The optional `capture_generated_tokens.py` command is the
188
+ explicit exception: it loads the pinned official generation runtime and uses the requested device (CUDA by default)
189
+ only to observe internally generated tokens. `pyproject.toml` requires `torch>=2.10.0`, after the vulnerable
190
+ versions listed in [GHSA-63cw-57p8-fm3p](https://github.com/advisories/GHSA-63cw-57p8-fm3p).
191
+
192
+ Install the optional generation-capture runtime with `pip install -e '.[capture]'`; that extra pins the audited
193
+ Diffusers Git revision and explicitly declares Accelerate, Hugging Face Hub, Transformers, and SoundFile in addition
194
+ to the base NumPy and Torch requirements.
195
+
196
+ ```bash
197
+ python inspect_dav.py /path/to/dav.pth --sha256
198
+ python inspect_dav.py /path/to/dav.pth --json
199
+
200
+ # Both commands return nonzero and say BLOCKED before opening missing.wav.
201
+ python encode_audio.py missing.wav --dav /path/to/dav.pth --json
202
+ python round_trip_test.py missing.wav reconstructed.wav --dav /path/to/dav.pth --json
203
+
204
+ # Exact comparison, only when genuine internal and re-encoded trajectories exist.
205
+ python native_token_compatibility.py internal.npy recovered.npy \
206
+ --reference-layout frames_first --recovered-layout frames_first --json
207
+
208
+ # Capture tokens from generation itself; this does not re-encode the saved WAV.
209
+ python capture_generated_tokens.py \
210
+ --prompt "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass." \
211
+ --lyrics "[instrumental]" --audio-duration 1 --num-inference-steps 30 --seed 7 \
212
+ --local-files-only \
213
+ --output-wav /path/to/music3_generated_1s.wav \
214
+ --output-tokens /path/to/music3_internal_tokens.npy \
215
+ --output-metadata /path/to/music3_generation_capture.json
216
+
217
+ # Generated-sample integration is enabled only when all four are set.
218
+ MINIMAX_DAV_PATH=/path/to/dav.pth \
219
+ MINIMAX_GENERATED_WAV_PATH=/path/to/music3_generated_1s.wav \
220
+ MINIMAX_INTERNAL_TOKENS_PATH=/path/to/music3_internal_tokens.npy \
221
+ MINIMAX_GENERATION_CAPTURE_PATH=/path/to/music3_generation_capture.json \
222
+ pytest
223
+ ```
224
+
225
+ Synthetic tests cover flat and nested/`62000_generator` checkpoints, capability verdicts, pre-WAV failure, explicit
226
+ token layouts, all codebook ranges, exact/mismatched agreement, unequal frame counts, capture ordering, metadata-last
227
+ replacement, source cleanliness, and partial artifact configuration. The environment-gated integration test checks
228
+ the real 548-tensor release, validates the full generated-capture schema, recomputes the WAV/NPY/source/script hashes,
229
+ and confirms the blocked encoder path. Impossible experiments--WAV token encoding, token round trip, re-encoding
230
+ agreement, and continuation--are explicitly reported as blocked rather than marked passed.
231
+
232
+ ## Audited sources and revisions
233
+
234
+ - MiniMax release checkpoint: [MiniMaxAI/MiniMax-Music3 at revision
235
+ `fbdf52fbaaca799592917417eb05f1899f1255ec`](https://huggingface.co/MiniMaxAI/MiniMax-Music3/tree/fbdf52fbaaca799592917417eb05f1899f1255ec).
236
+ - Hugging Face Diffusers revision `90b4e34e79a86ec5e7f2437634fe95ecd2108796`: the
237
+ [Music 3 converter's DAV selection](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/scripts/convert_minimax_music3_to_diffusers.py#L96-L127),
238
+ [global/local token generation and feedback embedding](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/src/diffusers/modular_pipelines/minimax_music3/encoders.py#L102-L148),
239
+ [25 Hz autoregressive loop](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/src/diffusers/modular_pipelines/minimax_music3/encoders.py#L287-L358), and
240
+ [seven-head local depth decoder](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/src/diffusers/models/transformers/minimax_music3_rvq_depth_decoder.py#L91-L139).
241
+ - SGLang-Omni revision `d0edde030334a1e54a6644ba3ab21eabc4a01f73`: exact
242
+ [special IDs and audio-code offset](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/prompt.py#L9-L20),
243
+ [eight-code feedback embedding](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/sglang_model.py#L82-L96),
244
+ [per-frame code/hidden alignment](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/model_runner.py#L252-L289), and
245
+ [decoder-only DAV selection](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/dav.py#L114-L154).
LICENSE ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or Derivative
95
+ Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work, excluding
103
+ those notices that do not pertain to any part of the Derivative
104
+ Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one of
111
+ the following places: within a NOTICE text file distributed as
112
+ part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and do
117
+ not modify the License. You may add Your own attribution notices
118
+ within Derivative Works that You distribute, alongside or as an
119
+ addendum to the NOTICE text from the Work, provided that such
120
+ additional attribution notices cannot be construed as modifying
121
+ the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright 2026 Music3Lab contributors
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
MODEL_CARD.md ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Model card — Music3Lab adapters & encoders
2
+
3
+ This card covers the trained checkpoints produced by Music3Lab. **The base
4
+ MiniMax-Music3 model is not covered here** — see its own card at
5
+ `MiniMaxAI/MiniMax-Music3`.
6
+
7
+ > ⚠️ **License:** these checkpoints are likely **derivative works of
8
+ > MiniMax-Music3** (trained through its frozen decoder and/or on its captured
9
+ > state). Their redistribution may be governed by the MiniMax-Music3 license.
10
+ > Confirm before publishing. See THIRD_PARTY.md. The checkpoints are **not** in
11
+ > this Git repo; only pointers are (`checkpoints/README.md`).
12
+
13
+ ## Common facts
14
+
15
+ - **Base model:** MiniMax-Music3 (frozen; never fine-tuned by this project).
16
+ - **What stays frozen:** the DAV/Flow decoder, the Global/Local language models,
17
+ and the vocoder. Only the small adapters/encoders below are trained.
18
+ - **Training compute:** single NVIDIA H100 80 GB.
19
+ - **Evaluation:** preregistered objective gates only (SI-SDR, SNR, correlation,
20
+ reconstruction ruler, latent NMSE, loudness/stereo, boundary continuity,
21
+ anti-copy). **No human listening tests. No learned aesthetic reward.**
22
+ - **Author's own songs** were held out of all training and checkpoint selection
23
+ (out-of-distribution evaluation only).
24
+
25
+ ## Checkpoints
26
+
27
+ | Checkpoint | Arch / trainable params | Trained on | Gate result |
28
+ |---|---|---|---|
29
+ | `flow-encoder` (base) | WAV `[B,2,44032]` → latent `[B,128,86]`, Conv1d frontend (~1M) | Music3-generated WAV/latent teacher pairs | ✅ pilot pass; ~1.34 ms one-pass |
30
+ | `external-finetune` encoder | same arch, fine-tuned | 542 LAION real-music + Music3 teachers | 🟡 **rejected specialist** — external held-out ruler +74.9%, SI-SDR 2.03→8.24 dB, but protected teacher latent NMSE regressed +13.2% (> 5% limit) |
31
+ | `masked-flow-inpaint` adapter | rank-4 LoRA on 108 QKV projs × 36 Flow blocks + 3 embeddings = 1,775,616 | captured Music3 conditions | ✅ pilot: +30.8% latent NMSE, +20.1% hole ruler (captured-condition only) |
32
+ | `learned-continuation` / `-residual` adapter | rank-4 Flow LoRA (~1.7M) | adjacent real-audio windows | ❌ lost to repeat/roll baselines |
33
+ | `native-token` adapter | 1×16384 semantic head + 7×1024 residual heads | 96 Music3 token captures | ❌ residual CE regressed; not a tokenizer |
34
+ | `acoustic-fim-v2` | 2-sided waveform U-Net, 4,063,090 | 542 LAION real-music | ❌ +5.9% vs +10% gate |
35
+ | `audio-prepend` / `waveform-right-context-prepend` | spectral/attention (~5.4M) | real-audio right-context | ❌ seam / anti-copy gates failed |
36
+ | `waveform-causal-continuation` | causal waveform net | real-audio history | ❌ collapsed to near-exact repeat-tail copy |
37
+ | `native-state-stage0` (v3) | 3-branch posterior (log-Mel + stereo STFT + latent), 8 Conformer blocks | one Music3 clip | ✅ one-clip 25/25 alignment (feedback NMSE 5.1e-13) — **interpolation, not a general encoder** |
38
+ | `native-state-distill` | c0 soft-logit distillation | 104 captured clips | 🟡 bounded pass (held-out hard CE 8.97); c1–c7 absent |
39
+ | `native-stage1-residual` | autoregressive c1–c7 heads | 1,024 captured clips | ❌ near-modal; 0 exact rows |
40
+ | `official_local_audio_heads` | frozen official Local heads, extracted | (extracted from base) | support artifact so native-state code runs without the 57 GB model |
41
+
42
+ Exact per-run commit hashes, selected steps, validation losses, and artifact
43
+ SHA-256s are in [`reports/RELEASE_STATUS.md`](reports/RELEASE_STATUS.md) and
44
+ [`reports/FINAL_RESULTS.md`](reports/FINAL_RESULTS.md).
45
+
46
+ ## Intended use
47
+
48
+ Research and reproduction: studying continuous-latent audio representations,
49
+ editing in Music3's Flow-latent space, capture/replay of generation state, and
50
+ objective evaluation methodology — including studying the **negative** results.
51
+
52
+ ## Out-of-scope / limitations
53
+
54
+ - Not a native WAV→token encoder for Music3 (that is unsolved with the released
55
+ weights).
56
+ - Continuous inversion is a slow research teacher (~208 s per 1 s), not real-time.
57
+ - Arbitrary-WAV continuation / inpainting / prepend **do not work** at the target
58
+ quality; those checkpoints are provided as reproducible negative results.
59
+ - No safety/aesthetic/musicality guarantees. Objective metrics ≠ musical quality.
NOTICE ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Music3Lab
2
+ Copyright 2026 Music3Lab contributors
3
+
4
+ This product includes software developed by the Music3Lab contributors.
5
+
6
+ Licensed under the Apache License, Version 2.0 (see LICENSE).
7
+
8
+ --------------------------------------------------------------------------------
9
+ This repository contains ONLY original source code, configs, scripts, tests,
10
+ documentation, and dataset *metadata* authored by the Music3Lab contributors.
11
+
12
+ It does NOT bundle, redistribute, or embed any of the following third-party
13
+ assets. These must be obtained by the user from their original sources under
14
+ their own licenses. See THIRD_PARTY.md for details.
15
+
16
+ - MiniMax-Music3 model weights (MiniMaxAI/MiniMax-Music3)
17
+ - The Hugging Face Diffusers library / its MiniMax-Music3 pipeline
18
+ - LAION-DISCO-12M audio (only Apache-2.0 metadata/IDs are referenced here;
19
+ no audio is included)
20
+ - Any YouTube-sourced audio
21
+ - Any third-party recordings
22
+
23
+ Trained adapter/encoder checkpoints produced by this project are distributed
24
+ separately (see checkpoints/README.md and MODEL_CARD.md) and may be subject to
25
+ the MiniMax-Music3 license as derivative artifacts. Confirm that license before
26
+ redistributing any checkpoint.
README.md ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Music3Lab
2
+
3
+ **Open research toolkit extending [MiniMax-Music3](https://huggingface.co/MiniMaxAI/MiniMax-Music3)
4
+ with arbitrary-audio encoding, continuation, inpainting, prepend generation,
5
+ prompt-free generation, and automated objective evaluation.**
6
+
7
+ Music3Lab is a reproducible, evidence-first toolkit built *around* the released
8
+ MiniMax-Music3 weights. It does not modify or redistribute those weights. Every
9
+ capability below was gated on preregistered objective metrics, and **the
10
+ failures are published alongside the successes** — they are the more useful part
11
+ of the research.
12
+
13
+ > **Honesty note.** This is a research toolkit, not a finished product. Several
14
+ > headline goals (native WAV→token encoding, true arbitrary-audio history-aware
15
+ > continuation/inpainting/prepend, direct long-form reference conditioning) were
16
+ > attempted and **did not pass their gates**. Those negative results, their code,
17
+ > configs, and exact metrics are all here on purpose.
18
+
19
+ ---
20
+
21
+ ## What it does
22
+
23
+ | Capability | Status | Notes / measured result |
24
+ |---|---|---|
25
+ | **Checkpoint audit** of the released weights | ✅ available | Every released tensor classified; proves there is **no** native RVQ waveform tokenizer in `dav.pth`. See [FINDINGS.md](FINDINGS.md). |
26
+ | **Continuous WAV → Flow-latent encoder** | ✅ available | One-pass ≈1.34 ms (base pilot). External real-music fine-tune improved held-out audio ruler 74.9% and SI-SDR 2.03→8.24 dB, but is a **rejected specialist** (protected teacher latent regressed +13.2%), not a promoted champion. |
27
+ | **Latent inversion** (research/oracle mode) | ✅ available | Iterative; a 1 s external clip reached 22.2 dB SI-SDR / 0.997 correlation. Slow (~208 s per 1 s) — a teacher, not a real-time encoder. |
28
+ | **Masked-Flow inpainting** (captured Music3 conditions) | ✅ available | +30.8% latent NMSE, +20.1% hole audio-ruler vs zero-adapter. Captured-condition only. |
29
+ | **Captured-state style continuation** | ✅ available | 12 s → 16 s, four candidates, objective style/seam ranking. Captured-state only, deterministic-from-frame-0. |
30
+ | **Full-state resume** | ✅ available | Serializes KV cache + CUDA RNG; reproduces frames + Flow chunks exactly across processes (deterministic backend). |
31
+ | **Reference-guided append** (CPU) | ✅ available | Appends a chosen reference-style candidate with **bit-exact** source preservation outside the crossfade. |
32
+ | **Prompt-free generation** | ✅ available | No user text; internal MIR/planning→text bridge. Best-of-N up to 90 s (max policy: 3/8 eligible full-length). |
33
+ | **Reference-style generation** | 🟡 partial | 8 s, ranked by direct continuous-latent + MIR similarity. Not direct model conditioning or full style transfer. |
34
+ | **Objective evaluation suite** | 🟡 partial | Integrity, reconstruction, SI-SDR/SNR, correlation, loudness/stereo, anti-copy. No learned musicality/aesthetic judges. |
35
+ | **Native WAV → Music3 RVQ tokens** | ❌ blocked | Released `dav.pth` has no quantizer/codebooks. (`62000_generator` is a PyTorch **ZIP folder name**, not a component.) |
36
+ | **Arbitrary-WAV continuation** | ❌ failed | Learned conditioner lost to repeat/roll baselines. |
37
+ | **Two-sided acoustic FIM (arbitrary WAV)** | ❌ failed | +5.9% ruler vs required +10%; boundaries worse than interpolation. |
38
+ | **Arbitrary-WAV / waveform prepend** | ❌ failed | Failed seam / anti-copy gates. |
39
+ | **Native-state Stage-1 residual prediction** | ❌ failed | Near-modal; mean CE 6.834, exact full token rows 0. |
40
+ | **Direct long-form reference conditioning** | ❌ failed | Tempo drift / early-EOS; no eligible 60 s candidate. |
41
+ | **Enforceable negative prompts (e.g. "no vocals")** | ❌ not enforceable | Reported honestly as `NOT_ENFORCEABLE`. |
42
+
43
+ `music3lab status --json` is the authoritative machine-readable capability
44
+ matrix. Full write-ups are in [`reports/`](reports/) and
45
+ [`reports/FINAL_RESULTS.md`](reports/FINAL_RESULTS.md).
46
+
47
+ ---
48
+
49
+ ## What it is NOT
50
+
51
+ - ❌ It does **not** include MiniMax-Music3 weights. You download those yourself.
52
+ - ❌ It does **not** redistribute any audio — not LAION/YouTube audio, not the
53
+ author's own songs. Only dataset *metadata* (IDs, hashes, splits) is included.
54
+ - ❌ It is **not** a native audio tokenizer for Music3. That does not exist in
55
+ the public release (see [FINDINGS.md](FINDINGS.md)).
56
+
57
+ ---
58
+
59
+ ## Quickstart
60
+
61
+ ```bash
62
+ # 1. Environment (Python 3.10; exact pins in requirements.lock)
63
+ python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
64
+ pip install -e . # core (CPU inspection/eval)
65
+ pip install -e ".[capture]" # + diffusers/transformers for generation (GPU)
66
+
67
+ # 2. Get the base model yourself (NOT bundled). See REPRODUCING.md.
68
+ hf download MiniMaxAI/MiniMax-Music3 --local-dir ./models/minimax-music3
69
+
70
+ # 3. Inspect the released checkpoint (CPU, no GPU, no weights modified)
71
+ python inspect_dav.py ./models/minimax-music3/dav.pth --json --sha256
72
+
73
+ # 4. Machine-readable capability matrix
74
+ music3lab status --json
75
+ ```
76
+
77
+ Full setup, model download, one-command demo, and benchmark commands:
78
+ [REPRODUCING.md](REPRODUCING.md).
79
+
80
+ ---
81
+
82
+ ## Repository layout
83
+
84
+ ```
85
+ music3lab/
86
+ ├── src/music3lab/ # the installable package (tested; layout preserved)
87
+ │ ├── codec/ # encoders, external fine-tune, native-state experiments
88
+ │ ├── editing/ # continuation / inpaint / prepend / append
89
+ │ ├── autonomous/ # champion/challenger promotion + rollback
90
+ │ ├── inversion*.py # latent inversion (research mode)
91
+ │ ├── eval.py # objective evaluation
92
+ │ └── ... # baseline capture, checkpoint audit, pipeline, release
93
+ ├── configs/ # frozen experiment/training configs (34)
94
+ ├── scripts/ # runnable training / experiment / benchmark scripts (38)
95
+ │ └── data/ # LAION downloader (laion_ingest.py, laion_freeze_interim.py)
96
+ ├── tests/ # focused + adversarial suites (58)
97
+ ├── reports/ # per-capability write-ups + FINAL_RESULTS.md
98
+ ├── data/laion_disco/ # dataset METADATA only (IDs, hashes, splits) — no audio
99
+ ├── checkpoints/ # POINTERS to trained adapters (no weights) — see README there
100
+ ├── examples/ # how to reproduce demo outputs (no bundled audio)
101
+ ├── requirements.lock # exact pinned environment
102
+ ├── LICENSE NOTICE THIRD_PARTY.md MODEL_CARD.md DATA.md TRAINING.md REPRODUCING.md
103
+ ```
104
+
105
+ **Training.** Every trainable component ships its training script + config +
106
+ tests. See [TRAINING.md](TRAINING.md) for the full table and how to push the
107
+ open problems (native tokenization, arbitrary-audio editing).
108
+
109
+ **Note on structure.** The conceptual grouping (encoder / continuation / inpaint
110
+ / prepend / eval) is preserved *thematically* via the `codec/` and `editing/`
111
+ subpackages and this map, rather than by physically splitting `src/` — that keeps
112
+ the 48-test suite green and the package importable for a reproducible v0.1.0. A
113
+ physical refactor into top-level `encoder/continuation/...` packages is a
114
+ possible later, separately-tested change.
115
+
116
+ ---
117
+
118
+ ## The core finding
119
+
120
+ The released `dav.pth` is a **continuous** DAV analysis encoder + Gaussian
121
+ posterior heads + flow model + waveform decoder. It contains **no** RVQ/VQ
122
+ quantizer, **no** acoustic codebooks, and **no** `generator`/`62000_generator`
123
+ tensors. The string `62000_generator` is only the root folder name inside the
124
+ PyTorch ZIP archive — not a model component. Music3's eight-stream token space
125
+ therefore cannot be produced from an arbitrary WAV with the released weights.
126
+ Everything Music3Lab does works either in the continuous Flow-latent space or
127
+ from *captured* generation state. Details and reproduction: [FINDINGS.md](FINDINGS.md).
128
+
129
+ ---
130
+
131
+ ## Trained checkpoints & data
132
+
133
+ - **Adapters/encoders** are released separately (Hugging Face) — see
134
+ [checkpoints/README.md](checkpoints/README.md) and [MODEL_CARD.md](MODEL_CARD.md).
135
+ ⚠️ They are derivatives of MiniMax-Music3 and may be governed by its license;
136
+ confirm before redistributing.
137
+ - **Dataset**: only LAION-DISCO-12M metadata + a downloader are shipped. No audio.
138
+ See [DATA.md](DATA.md).
139
+
140
+ ---
141
+
142
+ ## License
143
+
144
+ Original Music3Lab code: **Apache-2.0** ([LICENSE](LICENSE), [NOTICE](NOTICE)).
145
+ Third-party components and their separate licenses: [THIRD_PARTY.md](THIRD_PARTY.md).
REPORT_AUDIO_PREPEND.md ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Arbitrary-WAV suffix-conditioned prepend
2
+
3
+ This milestone trains a fresh, suffix-only continuous-latent projector and a
4
+ fresh rank-4 QKV Flow adapter on the frozen external 542/68/68 split. The
5
+ following one-second audio window is the only condition; the preceding window
6
+ is hidden and used only as the target. User-owned songs remain OOD evaluation
7
+ material and are not training inputs.
8
+
9
+ The output compositor is exactly:
10
+
11
+ `generated_prefix[:-T] + equal_power(generated_prefix[-T:], source[:T]) + source[T:]`
12
+
13
+ where `1 <= T <= 1024`. This is an acoustic local prepend experiment, not a
14
+ native-token, semantic, prompt-conditioned, or long-song prepend claim.
REPORT_REFERENCE_RANKED_POOL.md ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Reference-ranked retained long-form pool
2
+
3
+ Release status: **AVAILABLE_CPU_POST_GENERATION_ONLY** at implementation commit
4
+ `a7dd17852b0db6858aa32eb7a43c84c2e4c155f3`.
5
+
6
+ This milestone ranks only the three frozen, normalized exact-90-second retained
7
+ WAVs (`candidate-002`, `candidate-003`, and `candidate-007`) from the completed
8
+ prompt-free maximum-90-second pool. Raw renders, early-EOS rows, and unlisted
9
+ candidates are rejected before scoring.
10
+
11
+ The source WAV is not passed to a generator. A frozen rejected-specialist Flow
12
+ encoder recomputes deterministic eight-crop continuous-latent profiles on CPU,
13
+ and the f040 compatibility weights and hard gates select among existing WAVs.
14
+ No audio generation or training is performed. The runtime result is therefore
15
+ `reference_ranked_long_form_pool_not_direct_audio_conditioning`, not native style
16
+ transfer, direct audio conditioning, artistic-quality evidence, or enforceable
17
+ vocal exclusion.
18
+
19
+ The immutable run configuration is `configs/reference-ranked-pool-v1.yaml` and
20
+ the executable wrapper is `scripts/run_reference_ranked_pool.py`. The fresh
21
+ runtime JSON and Markdown report are published beside the selected WAV rather
22
+ than checked into source control.
23
+
24
+ Authoritative evidence is
25
+ `/home/ubuntu/minimax-reference-ranked-pool-evidence/loveonme-exact90-f040ab1`.
26
+ Selected WAV/manifest JSON/report SHA-256 values are
27
+ `d5cbd5c920745e106b14aaf6659d95255f8b0f8d73a8b9f97ab47ecc00f72207`,
28
+ `41725b020aadcae2d58461bd8c0ba19e00ae3ded1596d82df3e184fac9650b8b`,
29
+ and `44f174843c9103667b709dd41cd9d64158819b8721d183fcfe9cf6990be48075`.
30
+ Vocal exclusion is `NOT_ENFORCEABLE`; native negative prompting is false.
REPORT_REFERENCE_STYLE.md ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Reference-style candidate matching
2
+
3
+ `music3lab reference-style` generates eight independent Music3 candidates from
4
+ an internal text description derived only from measured tempo, key, energy, and
5
+ stereo width. A frozen external continuous encoder then scores deterministic
6
+ one-second crops from the reference and each candidate. The scorer is the
7
+ rejected external specialist and is explicitly restricted to
8
+ `REJECTED_EXTERNAL_SPECIALIST_STYLE_SCORER_ONLY`; it is neither retrained nor
9
+ promoted and does not produce Music3 native tokens.
10
+
11
+ Automatic selection combines direct latent-profile distance with measured MIR
12
+ compatibility and a bounded diversity ranking bonus. Hard technical and
13
+ anti-copy gates run before ranking. Creativity means candidate diversity only,
14
+ not artistic quality. This path is not direct model audio conditioning, full
15
+ semantic style transfer, guaranteed vocal exclusion, or artistic superiority.
16
+
17
+ ## Measured H100 result
18
+
19
+ At commit `f040ab152908ba343ea5e8ffbdb86fae05e4c211`, the path generated eight
20
+ independent eight-second candidates from the user reference and selected index
21
+ 0 / seed 101. The selected WAV SHA-256 is
22
+ `a4af7f05a6e22db7eb9696058f41b7a099f096415eb8212940d9bb9c58be795d`.
23
+ All eight candidates were scored using direct frozen continuous-latent
24
+ profiles, MIR compatibility, and a bounded creativity/diversity bonus.
25
+ Generation took 63.7525 seconds after one 4.8467-second pipeline load; peak
26
+ CUDA allocation was 24,467,387,392 bytes.
27
+
28
+ The `vocals` constraint was `NOT_ENFORCEABLE` and
29
+ `native_negative_prompt_used` was false. Evidence:
30
+ `/home/ubuntu/minimax-reference-style-evidence/user-loveonme-f040ab1-seed101.reference-style`
31
+ (manifest SHA-256
32
+ `2ada3a88bc57936810e7c8b8cf920d722225d438ca697745d9869ff4eeaa7485`).
REPRODUCING.md ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Reproducing Music3Lab
2
+
3
+ ## Hardware / software used
4
+
5
+ - **GPU:** 1× NVIDIA H100 80 GB (generation/training). CPU-only is enough for
6
+ checkpoint inspection, the objective evaluators, and all `preflight`/`status`
7
+ commands.
8
+ - **Driver:** 580.105.08 · **CUDA-capable** build of PyTorch.
9
+ - **OS:** Ubuntu 22.04 (Linux). The CLI also runs on Windows for CPU paths.
10
+ - **Python:** 3.10.12.
11
+
12
+ Exact pins are in [`requirements.lock`](requirements.lock) (75 packages) and
13
+ [`environment.yml`](environment.yml). Notable: `torch==2.13.0`,
14
+ `transformers==5.15.0`, `accelerate==1.14.0`, `safetensors==0.8.0`,
15
+ `numpy==2.2.6`, `pydantic==2.13.4`.
16
+
17
+ > **Diffusers is pinned to an exact commit**, not a PyPI release, because the
18
+ > MiniMax-Music3 pipeline landed on `main`:
19
+ > `diffusers @ git+https://github.com/huggingface/diffusers.git@90b4e34e79a86ec5e7f2437634fe95ecd2108796`
20
+ > (already declared in `pyproject.toml`'s `[capture]` extra).
21
+
22
+ ## 1. Install
23
+
24
+ ```bash
25
+ python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
26
+ pip install -e . # core: inspection + eval (CPU OK)
27
+ pip install -e ".[capture]" # + diffusers/transformers for generation (GPU)
28
+ # For a byte-exact environment instead:
29
+ # pip install -r requirements.lock
30
+ ```
31
+
32
+ ## 2. Get the base model (NOT bundled)
33
+
34
+ ```bash
35
+ hf download MiniMaxAI/MiniMax-Music3 --local-dir ./models/minimax-music3
36
+ ```
37
+
38
+ Music3Lab never modifies these files. `models/` is git-ignored.
39
+
40
+ ## 3. One-command checks (CPU, no weights modified)
41
+
42
+ ```bash
43
+ # The core finding: no RVQ tokenizer in the released checkpoint
44
+ python inspect_dav.py ./models/minimax-music3/dav.pth --json --sha256
45
+
46
+ # Machine-readable capability matrix (the source of truth for what works)
47
+ music3lab status --json
48
+
49
+ # Discover exact subcommands and their flags
50
+ music3lab --help
51
+ ```
52
+
53
+ ## 4. Demo generation (GPU + [capture] extra + downloaded model)
54
+
55
+ Use `music3lab --help` for the authoritative subcommand list and flags. The
56
+ demonstrated, passing generation/edit paths are:
57
+
58
+ - `music3lab` prompt-free best-of-N generation (up to 90 s)
59
+ - `music3lab` reference-style ranked generation (8 s)
60
+ - `music3lab captured-style-continue` (captured-state continuation, 12 s → 16 s)
61
+ - `music3lab reference-guided-append` (CPU append with exact source preservation)
62
+ - `music3lab reference-ranked-pool` (CPU ranking of a retained pool)
63
+
64
+ Each writes a WAV plus a JSON sidecar and a `report.md` with the objective
65
+ metrics and artifact SHA-256s. Preflight variants (`music3lab preflight ...`) do
66
+ static identity checks with **no GPU and no model load**.
67
+
68
+ ## 5. Benchmarks / evidence
69
+
70
+ Per-capability numbers, gates, selected steps, and artifact hashes are recorded
71
+ in [`reports/`](reports/) — start with
72
+ [`reports/RELEASE_STATUS.md`](reports/RELEASE_STATUS.md) and
73
+ [`reports/FINAL_RESULTS.md`](reports/FINAL_RESULTS.md). These are the frozen
74
+ measured results, not marketing claims.
75
+
76
+ ## 6. Regenerating training data / corpora
77
+
78
+ - Music3 teacher pairs (WAV/latent/token captures): `scripts/` capture runners,
79
+ which load the model you downloaded in step 2.
80
+ - External LAION corpus: `scripts/data/laion_ingest.py` then
81
+ `scripts/data/laion_freeze_interim.py` (adjust the hardcoded `ROOT` paths
82
+ first; see [DATA.md](DATA.md) and its ToS caveat).
83
+
84
+ ## Notes on determinism
85
+
86
+ Exact cross-process reproduction (full-state resume, captured-state continuation)
87
+ requires `torch.use_deterministic_algorithms(True)` **from frame 0**. The older
88
+ Phase-0 captures were produced under the nondeterministic SDPA backend and cannot
89
+ be bit-exactly resumed — a documented limitation, not a bug. See
90
+ [`docs/CAPTURED_STATE_STYLE_CONTINUATION.md`](docs/CAPTURED_STATE_STYLE_CONTINUATION.md).
capture_generated_tokens.py ADDED
@@ -0,0 +1,460 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Capture native token rows produced internally by official Music 3 generation.
2
+
3
+ This utility observes the existing autoregressive path. It does not encode or
4
+ re-encode a WAV and cannot turn arbitrary audio into Music 3 tokens.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import functools
11
+ import hashlib
12
+ import importlib
13
+ import json
14
+ import os
15
+ import subprocess
16
+ import tempfile
17
+ from contextlib import contextmanager
18
+ from dataclasses import dataclass, field
19
+ from pathlib import Path
20
+ from typing import Any, Callable, Iterator
21
+
22
+ import numpy as np
23
+
24
+ FRAME_RATE_HZ = 25
25
+ NUM_CODEBOOKS = 8
26
+ CODEBOOK_SIZES = (16_384, 1_024, 1_024, 1_024, 1_024, 1_024, 1_024, 1_024)
27
+
28
+
29
+ def _validate_token_rows(value: Any) -> np.ndarray:
30
+ array = np.asarray(value)
31
+ if array.ndim != 2 or array.shape[1] != NUM_CODEBOOKS:
32
+ raise ValueError(f"tokens must have shape [frames, 8], got {array.shape}")
33
+ if (
34
+ not np.issubdtype(array.dtype, np.integer)
35
+ or np.issubdtype(array.dtype, np.bool_)
36
+ ):
37
+ raise TypeError(f"tokens must use an integer dtype, got {array.dtype}")
38
+ if array.shape[0] == 0:
39
+ raise ValueError("token trajectory must contain at least one frame")
40
+ normalized = np.ascontiguousarray(array, dtype=np.int64)
41
+ for index, size in enumerate(CODEBOOK_SIZES):
42
+ values = normalized[:, index]
43
+ if int(values.min()) < 0 or int(values.max()) >= size:
44
+ raise ValueError(
45
+ f"codebook c{index} values must be in [0, {size - 1}]"
46
+ )
47
+ return normalized
48
+
49
+
50
+ DEFAULT_MODEL = "MiniMaxAI/MiniMax-Music3"
51
+ DEFAULT_MODEL_REVISION = "fbdf52fbaaca799592917417eb05f1899f1255ec"
52
+ DEFAULT_DIFFUSERS_REVISION = "90b4e34e79a86ec5e7f2437634fe95ecd2108796"
53
+
54
+
55
+ def _as_cpu_numpy(value: Any) -> np.ndarray:
56
+ if hasattr(value, "detach"):
57
+ value = value.detach()
58
+ if hasattr(value, "cpu"):
59
+ value = value.cpu()
60
+ if hasattr(value, "numpy"):
61
+ value = value.numpy()
62
+ return np.asarray(value)
63
+
64
+
65
+ @dataclass
66
+ class DepthCodeCapture:
67
+ """Record official depth-decoder outputs while preserving call behavior."""
68
+
69
+ priming_calls: int = 1
70
+ _rows: list[np.ndarray] = field(default_factory=list)
71
+
72
+ @property
73
+ def captured_calls(self) -> int:
74
+ return len(self._rows)
75
+
76
+ def wrap(self, original: Callable[..., Any]) -> Callable[..., Any]:
77
+ @functools.wraps(original)
78
+ def captured(*args: Any, **kwargs: Any) -> Any:
79
+ result = original(*args, **kwargs)
80
+ if not isinstance(result, (tuple, list)) or not result:
81
+ raise RuntimeError("official _generate_depth_codes returned no frame-code tensor")
82
+ paired = _as_cpu_numpy(result[0])
83
+ if paired.shape != (2, NUM_CODEBOOKS):
84
+ raise RuntimeError(
85
+ "official _generate_depth_codes must return paired [2, 8] codes, "
86
+ f"got {paired.shape}"
87
+ )
88
+ if not np.array_equal(paired[0], paired[1]):
89
+ raise RuntimeError("conditional/unconditional depth-code rows are not identical")
90
+ row = _validate_token_rows(paired[:1])[0]
91
+ self._rows.append(row.copy())
92
+ return result
93
+
94
+ return captured
95
+
96
+ def emitted_tokens(self) -> np.ndarray:
97
+ if self.priming_calls < 0:
98
+ raise ValueError("priming_calls must be non-negative")
99
+ if self.captured_calls <= self.priming_calls:
100
+ raise RuntimeError(
101
+ f"captured {self.captured_calls} call(s), not enough to skip "
102
+ f"{self.priming_calls} priming call(s)"
103
+ )
104
+ tokens = np.stack(self._rows[self.priming_calls :], axis=0)
105
+ return _validate_token_rows(tokens)
106
+
107
+
108
+ @contextmanager
109
+ def capture_official_depth_codes() -> Iterator[DepthCodeCapture]:
110
+ """Temporarily patch the official Diffusers depth-code function."""
111
+
112
+ encoders = importlib.import_module(
113
+ "diffusers.modular_pipelines.minimax_music3.encoders"
114
+ )
115
+ original = encoders._generate_depth_codes
116
+ capture = DepthCodeCapture(priming_calls=1)
117
+ patched = capture.wrap(original)
118
+ encoders._generate_depth_codes = patched
119
+ try:
120
+ yield capture
121
+ finally:
122
+ encoders._generate_depth_codes = original
123
+
124
+
125
+ def build_capture_metadata(
126
+ *,
127
+ tokens: Any,
128
+ audio: Any,
129
+ captured_calls: int,
130
+ priming_rows_skipped: int,
131
+ model_id: str,
132
+ requested_model_revision: str,
133
+ resolved_model_commit: str,
134
+ diffusers_identity: dict[str, Any],
135
+ device: str,
136
+ dtype: str,
137
+ prompt: str,
138
+ lyrics: str,
139
+ requested_audio_duration_seconds: float,
140
+ num_inference_steps: int,
141
+ seed: int,
142
+ sampling_rate: int,
143
+ wav_sha256: str,
144
+ tokens_npy_sha256: str,
145
+ capture_script_sha256: str,
146
+ ) -> dict[str, Any]:
147
+ """Validate and summarize one generated-audio/internal-token capture."""
148
+
149
+ normalized = _validate_token_rows(tokens)
150
+ waveform = _as_cpu_numpy(audio)
151
+ if waveform.ndim != 2 or waveform.shape[0] != 2:
152
+ raise ValueError(f"generated audio must have shape [2, samples], got {waveform.shape}")
153
+ if sampling_rate <= 0:
154
+ raise ValueError(f"sampling_rate must be positive, got {sampling_rate}")
155
+ if captured_calls != normalized.shape[0] + priming_rows_skipped:
156
+ raise ValueError(
157
+ "capture alignment mismatch: calls must equal emitted frames plus priming rows "
158
+ f"({captured_calls} != {normalized.shape[0]} + {priming_rows_skipped})"
159
+ )
160
+ if diffusers_identity.get("commit") is None:
161
+ raise ValueError("diffusers identity must include a commit")
162
+ for label, digest in (
163
+ ("WAV", wav_sha256),
164
+ ("NPY", tokens_npy_sha256),
165
+ ("capture script", capture_script_sha256),
166
+ ):
167
+ if len(digest) != 64 or any(char not in "0123456789abcdef" for char in digest):
168
+ raise ValueError(f"{label} SHA-256 must be 64 lowercase hexadecimal characters")
169
+ return {
170
+ "capture_kind": "official_internal_generation_tokens",
171
+ "wav_reencoding_performed": False,
172
+ "model_id": model_id,
173
+ "requested_model_revision": requested_model_revision,
174
+ "resolved_model_commit": resolved_model_commit,
175
+ "checkpoint_revision": resolved_model_commit,
176
+ "diffusers_revision": diffusers_identity["commit"],
177
+ "diffusers_source_identity": dict(diffusers_identity),
178
+ "capture_script_sha256": capture_script_sha256,
179
+ "device": device,
180
+ "dtype": dtype,
181
+ "prompt": prompt,
182
+ "lyrics": lyrics,
183
+ "requested_audio_duration_seconds": float(requested_audio_duration_seconds),
184
+ "num_inference_steps": int(num_inference_steps),
185
+ "seed": int(seed),
186
+ "captured_calls_including_priming": int(captured_calls),
187
+ "priming_rows_skipped": int(priming_rows_skipped),
188
+ "frame_rate_hz": FRAME_RATE_HZ,
189
+ "token_shape_frames_first": list(normalized.shape),
190
+ "token_mins": normalized.min(axis=0).tolist(),
191
+ "token_maxs": normalized.max(axis=0).tolist(),
192
+ "sampling_rate": int(sampling_rate),
193
+ "audio_shape_channels_first": list(waveform.shape),
194
+ "audio_duration_seconds": waveform.shape[1] / sampling_rate,
195
+ "artifact_sha256": {
196
+ "wav": wav_sha256,
197
+ "tokens_npy": tokens_npy_sha256,
198
+ },
199
+ }
200
+
201
+ def _sha256_file(path: Path) -> str:
202
+ digest = hashlib.sha256()
203
+ with path.open("rb") as handle:
204
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
205
+ digest.update(chunk)
206
+ return digest.hexdigest()
207
+
208
+
209
+ def _resolve_model_snapshot(
210
+ model_id: str,
211
+ requested_revision: str,
212
+ *,
213
+ local_files_only: bool,
214
+ ) -> tuple[Path, str]:
215
+ from huggingface_hub import hf_hub_download
216
+
217
+ # Resolving one required manifest is sufficient to bind a Hub revision to
218
+ # its immutable snapshot commit without requiring every optional repo file
219
+ # to have been cached by the component-selective official loader.
220
+ manifest = Path(
221
+ hf_hub_download(
222
+ repo_id=model_id,
223
+ filename="modular_model_index.json",
224
+ revision=requested_revision,
225
+ local_files_only=local_files_only,
226
+ )
227
+ ).absolute()
228
+ snapshot = manifest.parent
229
+ resolved_commit = snapshot.name
230
+ is_commit = len(requested_revision) == 40 and all(
231
+ char in "0123456789abcdefABCDEF" for char in requested_revision
232
+ )
233
+ if is_commit and resolved_commit.lower() != requested_revision.lower():
234
+ raise RuntimeError(
235
+ "model revision mismatch: "
236
+ f"requested {requested_revision}, resolved {resolved_commit}"
237
+ )
238
+ return snapshot, resolved_commit
239
+
240
+ def _verified_diffusers_revision(
241
+ diffusers_module: Any, expected: str
242
+ ) -> dict[str, Any]:
243
+ source = Path(diffusers_module.__file__).resolve()
244
+ roots = [source.parent, *source.parents]
245
+ repository = next((root for root in roots if (root / ".git").exists()), None)
246
+ actual = None
247
+ if repository is not None:
248
+ completed = subprocess.run(
249
+ ["git", "-C", str(repository), "rev-parse", "HEAD"],
250
+ check=True,
251
+ capture_output=True,
252
+ text=True,
253
+ )
254
+ actual = completed.stdout.strip()
255
+ status = subprocess.run(
256
+ [
257
+ "git",
258
+ "-C",
259
+ str(repository),
260
+ "status",
261
+ "--porcelain",
262
+ "--untracked-files=no",
263
+ ],
264
+ check=True,
265
+ capture_output=True,
266
+ text=True,
267
+ ).stdout.strip()
268
+ if status:
269
+ raise RuntimeError(
270
+ "Diffusers source checkout has dirty tracked files; "
271
+ "refusing an unverifiable capture"
272
+ )
273
+ source_kind = "git_checkout"
274
+ source_location = repository
275
+ tracked_clean: bool | None = True
276
+ else:
277
+ from importlib import metadata as importlib_metadata
278
+
279
+ direct_url = importlib_metadata.distribution("diffusers").read_text(
280
+ "direct_url.json"
281
+ )
282
+ if direct_url:
283
+ actual = json.loads(direct_url).get("vcs_info", {}).get("commit_id")
284
+ source_kind = "vcs_install"
285
+ source_location = source.parent
286
+ tracked_clean = None
287
+ if actual is None:
288
+ raise RuntimeError(f"cannot verify Diffusers git revision from {source}")
289
+ if actual != expected:
290
+ raise RuntimeError(f"Diffusers revision mismatch: expected {expected}, got {actual}")
291
+ relevant_source = (
292
+ source.parent / "modular_pipelines" / "minimax_music3" / "encoders.py"
293
+ )
294
+ if not relevant_source.is_file():
295
+ raise RuntimeError(
296
+ f"cannot hash relevant Diffusers Music 3 source: {relevant_source}"
297
+ )
298
+ return {
299
+ "commit": actual,
300
+ "source_kind": source_kind,
301
+ "source_location": str(source_location),
302
+ "tracked_clean": tracked_clean,
303
+ "relevant_source_path": str(relevant_source),
304
+ "relevant_source_sha256": _sha256_file(relevant_source),
305
+ }
306
+
307
+
308
+ def _temporary_sibling(target: Path) -> Path:
309
+ target.parent.mkdir(parents=True, exist_ok=True)
310
+ descriptor, name = tempfile.mkstemp(
311
+ prefix=f".{target.name}.",
312
+ suffix=".tmp",
313
+ dir=target.parent,
314
+ )
315
+ os.close(descriptor)
316
+ return Path(name)
317
+
318
+
319
+ def _fsync_file(path: Path) -> None:
320
+ with path.open("rb") as handle:
321
+ os.fsync(handle.fileno())
322
+
323
+ def run_capture(args: argparse.Namespace) -> dict[str, Any]:
324
+ """Load the pinned official runtime, generate audio, and save its internal codes."""
325
+
326
+ # Heavy runtime dependencies remain lazy so importing the pure capture
327
+ # invariants does not initialize a model or accelerator.
328
+ import soundfile as sf
329
+ import torch
330
+ import diffusers
331
+ from diffusers import ModularPipeline
332
+
333
+ diffusers_identity = _verified_diffusers_revision(
334
+ diffusers, args.diffusers_revision
335
+ )
336
+ snapshot_path, resolved_model_commit = _resolve_model_snapshot(
337
+ args.model,
338
+ args.model_revision,
339
+ local_files_only=args.local_files_only,
340
+ )
341
+ dtype = getattr(torch, args.dtype)
342
+ pipe = ModularPipeline.from_pretrained(str(snapshot_path))
343
+ pipe.load_components(dtype=dtype)
344
+ actual_frame_rate = float(pipe.frame_rate)
345
+ if actual_frame_rate != FRAME_RATE_HZ:
346
+ raise RuntimeError(
347
+ f"Music 3 frame-rate mismatch: expected {FRAME_RATE_HZ}, got {actual_frame_rate}"
348
+ )
349
+ pipe.to(args.device)
350
+ generator = torch.Generator(args.device).manual_seed(args.seed)
351
+
352
+ with capture_official_depth_codes() as capture:
353
+ audios = pipe(
354
+ prompt=args.prompt,
355
+ lyrics=args.lyrics,
356
+ audio_duration=args.audio_duration,
357
+ num_inference_steps=args.num_inference_steps,
358
+ generator=generator,
359
+ output="audios",
360
+ )
361
+
362
+ tokens = capture.emitted_tokens()
363
+ audio = _as_cpu_numpy(audios[0]).astype(np.float32, copy=False)
364
+ output_tokens = Path(args.output_tokens).expanduser().resolve()
365
+ output_wav = Path(args.output_wav).expanduser().resolve()
366
+ output_metadata = Path(args.output_metadata).expanduser().resolve()
367
+ temp_tokens = _temporary_sibling(output_tokens)
368
+ temp_wav = _temporary_sibling(output_wav)
369
+ temp_metadata = _temporary_sibling(output_metadata)
370
+ try:
371
+ with temp_tokens.open("wb") as handle:
372
+ np.save(handle, tokens)
373
+ handle.flush()
374
+ os.fsync(handle.fileno())
375
+ sf.write(
376
+ temp_wav,
377
+ audio.T,
378
+ int(pipe.sampling_rate),
379
+ format="WAV",
380
+ )
381
+ _fsync_file(temp_wav)
382
+ wav_sha256 = _sha256_file(temp_wav)
383
+ tokens_npy_sha256 = _sha256_file(temp_tokens)
384
+ metadata = build_capture_metadata(
385
+ tokens=tokens,
386
+ audio=audio,
387
+ captured_calls=capture.captured_calls,
388
+ priming_rows_skipped=capture.priming_calls,
389
+ model_id=args.model,
390
+ requested_model_revision=args.model_revision,
391
+ resolved_model_commit=resolved_model_commit,
392
+ diffusers_identity=diffusers_identity,
393
+ device=args.device,
394
+ dtype=args.dtype,
395
+ prompt=args.prompt,
396
+ lyrics=args.lyrics,
397
+ requested_audio_duration_seconds=args.audio_duration,
398
+ num_inference_steps=args.num_inference_steps,
399
+ seed=args.seed,
400
+ sampling_rate=int(pipe.sampling_rate),
401
+ wav_sha256=wav_sha256,
402
+ tokens_npy_sha256=tokens_npy_sha256,
403
+ capture_script_sha256=_sha256_file(Path(__file__).resolve()),
404
+ )
405
+ with temp_metadata.open("w", encoding="utf-8") as handle:
406
+ json.dump(metadata, handle, indent=2)
407
+ handle.write("\n")
408
+ handle.flush()
409
+ os.fsync(handle.fileno())
410
+
411
+ # Metadata is the commit record for the pair and is replaced last. If
412
+ # an earlier replacement is interrupted, its hashes cannot validate.
413
+ os.replace(temp_tokens, output_tokens)
414
+ os.replace(temp_wav, output_wav)
415
+ os.replace(temp_metadata, output_metadata)
416
+ return metadata
417
+ finally:
418
+ for temporary in (temp_tokens, temp_wav, temp_metadata):
419
+ temporary.unlink(missing_ok=True)
420
+
421
+ def build_parser() -> argparse.ArgumentParser:
422
+ parser = argparse.ArgumentParser(description=__doc__)
423
+ parser.add_argument("--prompt", required=True)
424
+ lyrics = parser.add_mutually_exclusive_group(required=True)
425
+ lyrics.add_argument("--lyrics")
426
+ lyrics.add_argument("--lyrics-file")
427
+ parser.add_argument("--audio-duration", type=float, default=1.0)
428
+ parser.add_argument("--num-inference-steps", type=int, default=30)
429
+ parser.add_argument("--seed", type=int, default=7)
430
+ parser.add_argument("--model", default=DEFAULT_MODEL)
431
+ parser.add_argument("--model-revision", default=DEFAULT_MODEL_REVISION)
432
+ parser.add_argument("--diffusers-revision", default=DEFAULT_DIFFUSERS_REVISION)
433
+ parser.add_argument(
434
+ "--local-files-only",
435
+ action="store_true",
436
+ help="resolve the pinned model revision from the local Hub cache only",
437
+ )
438
+ parser.add_argument("--device", default="cuda")
439
+ parser.add_argument(
440
+ "--dtype",
441
+ choices=("bfloat16", "float16", "float32"),
442
+ default="bfloat16",
443
+ )
444
+ parser.add_argument("--output-wav", default="music3_generated.wav")
445
+ parser.add_argument("--output-tokens", default="music3_internal_tokens.npy")
446
+ parser.add_argument("--output-metadata", default="music3_generation_capture.json")
447
+ return parser
448
+
449
+
450
+ def main(argv: list[str] | None = None) -> int:
451
+ args = build_parser().parse_args(argv)
452
+ if args.lyrics_file:
453
+ args.lyrics = Path(args.lyrics_file).expanduser().read_text(encoding="utf-8")
454
+ metadata = run_capture(args)
455
+ print(json.dumps(metadata, indent=2))
456
+ return 0
457
+
458
+
459
+ if __name__ == "__main__":
460
+ raise SystemExit(main())
checkpoints/README.md ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Checkpoints (pointers only — no weights in Git)
2
+
3
+ This directory intentionally contains **no model weights**. Trained
4
+ adapters/encoders are distributed separately (Hugging Face) and are pulled in by
5
+ you. See [../MODEL_CARD.md](../MODEL_CARD.md) for what each one is and its gate
6
+ result.
7
+
8
+ > ⚠️ **Before any of these are uploaded/redistributed:** confirm the
9
+ > MiniMax-Music3 license permits derivative-weight redistribution (see
10
+ > [../THIRD_PARTY.md](../THIRD_PARTY.md)). These checkpoints were trained through
11
+ > the frozen Music3 decoder / on captured Music3 state and may be governed by it.
12
+
13
+ ## Expected layout once downloaded
14
+
15
+ Place downloaded checkpoints here (git-ignored) or point configs at them:
16
+
17
+ ```
18
+ checkpoints/
19
+ ├── flow-encoder/encoder.safetensors # base continuous encoder (~1.34 ms)
20
+ ├── external-finetune/encoder.safetensors # rejected real-music specialist
21
+ ├── masked-flow-inpaint/adapter.safetensors # captured-condition inpaint pilot
22
+ ├── native-state-stage0-v3/native_state_stage0.safetensors
23
+ ├── native-state-distill/native_state_distilled.safetensors
24
+ └── ... # failed-experiment adapters (see MODEL_CARD)
25
+ ```
26
+
27
+ ## Downloading (once the HF repo exists and license is confirmed)
28
+
29
+ ```bash
30
+ hf download <your-org>/music3lab-adapters --local-dir ./checkpoints
31
+ ```
32
+
33
+ `<your-org>/music3lab-adapters` is a placeholder — replace with the real
34
+ Hugging Face repo id after the license review. Until then, this directory is a
35
+ pointer only.
configs/acoustic-fim-v2.yaml ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.acoustic-fim-v2-config.v1
2
+ capability: arbitrary_waveform_short_acoustic_fim_not_native_tokens_not_semantic_fim
3
+ seed: 260816
4
+ steps: 12000
5
+ validation_interval: 500
6
+ sample_rate: 44100
7
+ samples_per_frame: 512
8
+ canvas_samples: 98304
9
+ left_context_samples: 32768
10
+ right_context_samples: 32768
11
+ source_endpoint_guard_samples: 220500
12
+ hole_frames: [16, 32]
13
+ model_channels: [48, 64, 96, 128, 160, 192]
14
+ bottleneck_dilations: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512]
15
+ benchmark_batch_sizes_h100: [24, 32]
16
+ batch_24gb: 4
17
+ accumulation_24gb: 6
18
+ max_lr: 0.0002
19
+ min_lr: 0.00002
20
+ weight_decay: 0.01
21
+ gradient_clip_norm: 1.0
22
+ validation_namespace: music3lab.acoustic-fim-v2.validation.v1
23
+ heldout_namespace: music3lab.acoustic-fim-v2.heldout.v1
24
+ manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
25
+ splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
26
+ loss_weights:
27
+ hole_l1: 1.0
28
+ hole_l2: 0.5
29
+ mrstft: 0.25
30
+ complex_stft_nmse: 0.15
31
+ mid_side: 0.10
32
+ boundary_waveform_first_difference: 0.20
33
+ nonsilent_rms_floor: 0.0001
34
+ noncopy_correlation_threshold: 0.995
35
+ resampler: scipy.signal.resample_poly Kaiser-5.0 deterministic CPU
36
+ augmentations:
37
+ gain_db: [-3.0, 3.0]
38
+ stereo_balance_db: [-1.5, 1.5]
39
+ shared_polarity_probability: 0.5
40
+ channel_swap_probability: 0.5
41
+ forbidden: [side_shift, noise, resample]
configs/audio-continuation-v1.yaml ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.audio-continuation-config.v1
2
+ capability: continuous-latent local continuation; not native AR/token continuation or long-song generalization
3
+ sample_rate: 44100
4
+ source_samples: 44032
5
+ append_samples: 44032
6
+ overlap_samples: 1024
7
+ seed: 101
8
+ condition_repeat: 16
9
+ condition_weight: 0.95
10
+ temporal_context_weight: 0.05
11
+ fixed_noise_weight: 0.005
12
+ unrelated_condition: reverse_frames_and_features
13
+ reference_metric: frozen_five_term_audio_ruler_lower_is_better
14
+ continuity_metric: join_edge_rms_log_error_lower_is_better
15
+ zero_condition_baseline: true
16
+ unrelated_condition_baseline: true
17
+ strict_improvement_over_both_baselines: true
18
+ baseline_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
19
+ specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
20
+ specialist_source_gate: REJECTED_teacher_latent_nmse_regression
21
+ specialist_champion_eligible: false
22
+ text_input_allowed: false
23
+ captured_condition_allowed: false
24
+ native_tokens_allowed: false
25
+ future_audio_allowed: false
configs/audio-fim-v1.yaml ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.audio-fim-config.v1
2
+ steps: 1200
3
+ validation_interval: 100
4
+ euler_steps: 30
5
+ guidance_scale: 1.7
6
+ seed: 260816
7
+ noise_seed: 84119
8
+ benchmark_batch_sizes: [4, 8, 16]
9
+ max_lr: 0.00005
10
+ min_lr: 0.000005
11
+ hole_frames: [16, 32]
12
+ minimum_context_frames_each_side: 16
13
+ sample_rate: 44100
14
+ samples_per_frame: 512
15
+ external_encoder_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
16
+ external_encoder_role: REJECTED_EXTERNAL_SPECIALIST_NOT_CHAMPION
17
+ corpus_manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
18
+ corpus_splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
configs/audio-prepend-v1.yaml ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.audio-prepend-config.v1
2
+ steps: 1200
3
+ validation_interval: 100
4
+ euler_steps: 30
5
+ guidance_scale: 1.7
6
+ seed: 260816
7
+ noise_seed: 94119
8
+ benchmark_batch_sizes: [4, 8, 16]
9
+ max_lr: 0.00005
10
+ min_lr: 0.000005
11
+ transition_samples: 1024
12
+ sample_rate: 44100
13
+ samples_per_window: 44032
14
+ baselines: [zero_condition, unrelated_suffix, repeat_future, roll_future, silence]
15
+ external_encoder_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
16
+ external_encoder_role: REJECTED_EXTERNAL_SPECIALIST_NOT_CHAMPION
17
+ corpus_manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
18
+ corpus_splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
configs/autonomous-interim678-v1.yaml ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.autonomous-round.interim678.v1
2
+ capability: continuous_flow_encoder_reconstruction_only_not_native_tokens
3
+ suite_label: sealed_frozen_not_secret
4
+ split_seed: 16082026
5
+ bootstrap_seed: 41
6
+ bootstrap_samples: 2000
7
+ teacher_latent_nmse_maximum_regression_fraction: 0.05
8
+ protected_regression_maximum_fraction: 0.0
9
+ technical:
10
+ finite_required: true
11
+ maximum_absolute_audio: 1.0
12
+ selection_split: validation
13
+ promotion_splits: [public, sealed]
14
+ user22_excluded: true
15
+ training: false
16
+ candidates:
17
+ champion_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
18
+ specialist_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
19
+ zero_latent: synthetic_control
20
+ train_lookup: synthetic_control_forbidden_from_promotion
configs/baseline.yaml ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.baseline-config.v3
2
+ model_id: MiniMaxAI/MiniMax-Music3
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ diffusers_source_root: ../diffusers-music3
7
+ frozen_base_root: models/base/minimax-music3
8
+ frozen_manifest_root: manifests/base
9
+ artifacts_root: artifacts/baseline
10
+ report_path: reports/BASELINE.md
11
+ device: cuda
12
+ dtype: bfloat16
13
+ shard_budget_mib: 128
14
+ cases:
15
+ - name: short_parity
16
+ prompt:
17
+ provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
18
+ inline: "Genre: acoustic pop. BPM: 96. Key: C major. Warm and intimate, building gently into the chorus. Vocals: soft female lead, close and breathy, light stacked harmonies in the chorus. Arrangement: fingerpicked guitar and soft piano; brushed drums and upright bass enter in the chorus."
19
+ lyrics:
20
+ provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
21
+ inline: |-
22
+ [verse]
23
+ Morning light filtering through the pine
24
+ Every quiet street is yours and mine
25
+ [chorus]
26
+ Softly the world begins to breathe
27
+ audio_duration_seconds: 1.0
28
+ num_inference_steps: 30
29
+ ar_seed: 7
30
+ rng_policy: shared_stock
31
+ guidance_scale: 1.7
32
+ instrumented: false
33
+ requires_stock_capture_parity: true
34
+ - name: multichunk_acceptance
35
+ prompt:
36
+ provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
37
+ inline: "Genre: acoustic pop. BPM: 96. Key: C major. Warm and intimate, building gently into the chorus. Vocals: soft female lead, close and breathy, light stacked harmonies in the chorus. Arrangement: fingerpicked guitar and soft piano; brushed drums and upright bass enter in the chorus."
38
+ lyrics:
39
+ provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
40
+ inline: |-
41
+ [verse]
42
+ Morning light filtering through the pine
43
+ Every quiet street is yours and mine
44
+ [chorus]
45
+ Softly the world begins to breathe
46
+ audio_duration_seconds: 12.0
47
+ num_inference_steps: 30
48
+ ar_seed: 7
49
+ rng_policy: shared_stock
50
+ guidance_scale: 1.7
51
+ instrumented: true
52
+ requires_multichunk_acceptance: true
configs/captured-state-style-continuation-v1.yaml ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.captured-state-style-continuation.v1
2
+ capability: captured_state_same_style_continuation_only
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
5
+ base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
6
+ execution_policy: production_deterministic_from_frame0
7
+ dtype: bfloat16
8
+ root_seed: 7007
9
+ candidates: 4
10
+ prefix_frames: 300
11
+ new_frames: 100
12
+ sample_rate: 44100
13
+ source_samples: 529200
14
+ output_samples: 705600
15
+ guard_samples: 2048
16
+ flow_chunk_starts: [0, 100, 200]
17
+ inference_steps: 30
18
+ checkpoint_state_sha256: 848fff52c474b97aac46616cdaf4f67370212c95872fa4a660220689c7789f80
19
+ prefix_fused_bundle_sha256: 15cfcd31d0561a60a7e7a5fb04d3bb3d4889d41a6372dcdf7a0cf13256a3b01e
20
+ specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
21
+ ranking_weights:
22
+ latent: 0.42
23
+ tempo: 0.18
24
+ chroma: 0.10
25
+ energy: 0.10
26
+ stereo: 0.10
27
+ spectral: 0.10
28
+ seam_penalty_weight: 0.10
29
+ creativity_bonus_weight: 0.10
30
+ selection_tiebreak: candidate_index
31
+ scoring_region: extension_only[529200:705600]
32
+ arbitrary_wav: false
33
+ native_tokenizer: false
34
+ semantic_style_transfer: false
35
+ historical_phase0_rng: false
36
+ masked_flow_cleanup: false
configs/continuation-tier-a.yaml ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.captured-continuation-config.v2
2
+ capability: fresh_same_process_captured_prompt_continuation
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ dtype: bfloat16
7
+ device: cuda
8
+ append_frames: 25
9
+ num_inference_steps: 30
10
+ equal_power_overlap_samples: 11008
11
+ require_exact_appended_frames: true
12
+ cases:
13
+ - name: short_captured_continuation
14
+ phase0_case_name: short_parity
15
+ source_run_id: b0ee35156dad797b7d120419782d5c1d8847f24ae8c5c3aa3d0bd2e1dd7baa6d
16
+ continue_from_cache: true
17
+ ar_seed: 7007
18
+ - name: multichunk_replay
19
+ phase0_case_name: multichunk_acceptance
20
+ source_run_id: 2224b993a75741fb7391b545cb8687c14a131aec4568925bec3dd565ca84a704
21
+ continue_from_cache: false
22
+ ar_seed: 7013
23
+
24
+ # Tier-A requires authenticated native token/fused/flow capture state.
25
+ # Arbitrary-WAV continuation remains BLOCKED.
configs/external-flow-encoder-interim-678-v1.yaml ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.external-flow-encoder-finetune.interim678.v1
2
+ capability: external_waveform_to_continuous_flow_latents_not_native_rvq
3
+ initial_flow_config_file_sha256: 4a2474a50dd1e9e93c15cb5cf792e344531809a868b8f1b7405846a6012b47df
4
+ initial_flow_config_semantic_digest: 7b1ab53def143eca51e37af6089b28596f814ecdd11fc12312f8d76f05f13db3
5
+ initial_encoder_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
6
+ data:
7
+ accepted_count: 678
8
+ train_count: 542
9
+ validation_count: 68
10
+ heldout_count: 68
11
+ user22_count: 22
12
+ split_seed: 73
13
+ sample_rate: 44100
14
+ channels: 2
15
+ crop_samples: 44032
16
+ end_guard_seconds: 5
17
+ tranche_label: interim_tranche_678_due_systemic_bot_auth
18
+ dataset_status: not_2k_final
19
+ original_target_count: 2000
20
+ manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
21
+ splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
22
+ corpus_semantic_digest: 6cd403e9cd25e61d062af6caa9c6b136b751c7f668f1af7d0972989963ac3452
23
+ user22_hash_manifest_sha256: 924804bffd3c7cbc5d13f438a1a9ce8024842bfc0898504bc084af35d564a41a
24
+ training:
25
+ steps: 12000
26
+ batch_size: 8
27
+ external_batch_size: 6
28
+ music3_batch_size: 2
29
+ seed: 7302026
30
+ warmup_steps: 500
31
+ maximum_learning_rate: 0.0001
32
+ minimum_learning_rate: 0.00001
33
+ beta1: 0.9
34
+ beta2: 0.999
35
+ epsilon: 0.00000001
36
+ weight_decay: 0.0001
37
+ gradient_clip_norm: 1.0
38
+ validation_interval_steps: 500
39
+ autocast_dtype: bfloat16
40
+ master_parameter_dtype: float32
41
+ loss:
42
+ external_ruler_weight: 0.70
43
+ teacher_weight: 0.30
44
+ teacher_latent_nmse_weight: 1.0
45
+ teacher_ruler_weight: 0.05
46
+ external_prior_weight: 0.005
47
+ prior_latent_mean: 0.07592402398586273
48
+ prior_latent_std: 2.206683874130249
49
+ prior_tail_threshold: 6.0
50
+ prior_tail_weight: 0.1
51
+ evaluation:
52
+ minimum_external_ruler_improvement_fraction: 0.10
53
+ minimum_external_si_sdr_improvement_db: 1.0
54
+ minimum_external_correlation_improvement: 0.03
55
+ maximum_teacher_latent_nmse_regression_fraction: 0.05
56
+ maximum_teacher_ruler_regression_fraction: 0.05
57
+ maximum_one_pass_latency_ms: 2.0
58
+ user22_report_after_checkpoint_lock: true
59
+ selection_split: validation
60
+ frozen_vocoder: true
61
+ user22_selection_forbidden: true
62
+ native_tokenizer: false
63
+ generalization_claim: false
configs/external-flow-encoder-v1.yaml ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.external-flow-encoder-finetune.v1
2
+ capability: external_waveform_to_continuous_flow_latents_not_native_rvq
3
+ initial_flow_config_file_sha256: 4a2474a50dd1e9e93c15cb5cf792e344531809a868b8f1b7405846a6012b47df
4
+ initial_flow_config_semantic_digest: 7b1ab53def143eca51e37af6089b28596f814ecdd11fc12312f8d76f05f13db3
5
+ initial_encoder_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
6
+ data:
7
+ accepted_count: 2000
8
+ train_count: 1600
9
+ validation_count: 200
10
+ heldout_count: 200
11
+ user22_count: 22
12
+ split_seed: 73
13
+ sample_rate: 44100
14
+ channels: 2
15
+ crop_samples: 44032
16
+ end_guard_seconds: 5
17
+ training:
18
+ steps: 12000
19
+ batch_size: 8
20
+ external_batch_size: 6
21
+ music3_batch_size: 2
22
+ seed: 7302026
23
+ warmup_steps: 500
24
+ maximum_learning_rate: 0.0001
25
+ minimum_learning_rate: 0.00001
26
+ beta1: 0.9
27
+ beta2: 0.999
28
+ epsilon: 0.00000001
29
+ weight_decay: 0.0001
30
+ gradient_clip_norm: 1.0
31
+ validation_interval_steps: 500
32
+ autocast_dtype: bfloat16
33
+ master_parameter_dtype: float32
34
+ loss:
35
+ external_ruler_weight: 0.70
36
+ teacher_weight: 0.30
37
+ teacher_latent_nmse_weight: 1.0
38
+ teacher_ruler_weight: 0.05
39
+ external_prior_weight: 0.005
40
+ prior_latent_mean: 0.07592402398586273
41
+ prior_latent_std: 2.206683874130249
42
+ prior_tail_threshold: 6.0
43
+ prior_tail_weight: 0.1
44
+ evaluation:
45
+ minimum_external_ruler_improvement_fraction: 0.10
46
+ minimum_external_si_sdr_improvement_db: 1.0
47
+ minimum_external_correlation_improvement: 0.03
48
+ maximum_teacher_latent_nmse_regression_fraction: 0.05
49
+ maximum_teacher_ruler_regression_fraction: 0.05
50
+ maximum_one_pass_latency_ms: 2.0
51
+ user22_report_after_checkpoint_lock: true
52
+ selection_split: validation
53
+ frozen_vocoder: true
54
+ user22_selection_forbidden: true
55
+ native_tokenizer: false
56
+ generalization_claim: false
configs/flow-encoder-v1.yaml ADDED
@@ -0,0 +1,78 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.flow-encoder-config.v1
2
+ capability: waveform_to_continuous_flow_vocoder_latents_not_native_rvq
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ teacher:
7
+ prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
8
+ lyrics: "[instrumental]"
9
+ duration_seconds: 1.0
10
+ inference_steps: 12
11
+ guidance_scale: 1.7
12
+ train_seed_start: 1000
13
+ train_count: 64
14
+ validation_seed_start: 2000
15
+ validation_count: 16
16
+ heldout_seed_start: 3000
17
+ heldout_count: 16
18
+ model:
19
+ sample_rate: 44100
20
+ input_channels: 2
21
+ input_samples: 44032
22
+ latent_channels: 128
23
+ latent_frames: 86
24
+ frontend_kernel: 1024
25
+ frontend_stride: 512
26
+ frontend_padding: 256
27
+ width: 192
28
+ group_norm_groups: 24
29
+ residual_dilations: [1, 3, 9, 27]
30
+ training:
31
+ steps: 2000
32
+ batch_size: 8
33
+ seed: 424242
34
+ maximum_learning_rate: 0.0003
35
+ minimum_learning_rate: 0.00003
36
+ beta1: 0.9
37
+ beta2: 0.999
38
+ epsilon: 0.00000001
39
+ weight_decay: 0.0001
40
+ gradient_clip_norm: 1.0
41
+ validation_interval_steps: 100
42
+ latent_loss_weight: 1.0
43
+ audio_ruler_weight: 0.05
44
+ master_parameter_dtype: float32
45
+ autocast_dtype: bfloat16
46
+ decoder_dtype: bfloat16
47
+ loss:
48
+ latent_mean: 0.07592402398586273
49
+ latent_std: 2.206683874130249
50
+ stft_center: false
51
+ stft_fft_sizes: [256, 512, 1024, 2048]
52
+ stft_hop_sizes: [64, 128, 256, 512]
53
+ time_denominator_epsilon: 0.00000001
54
+ complex_stft_denominator_epsilon: 0.00000001
55
+ legacy_mrstft_epsilon: 0.0000001
56
+ mid_side_floor_fraction: 0.0001
57
+ envelope_windows: [63, 255, 1023]
58
+ relative_envelope_denominator_epsilon: 0.00000001
59
+ ruler_weights:
60
+ time_nmse: 2.0
61
+ complex_stft_nmse: 1.0
62
+ legacy_mrstft: 0.05
63
+ mid_side_nmse: 0.5
64
+ relative_envelope: 0.02
65
+ evaluation:
66
+ refinement_steps: 20
67
+ refinement_learning_rate: 0.01
68
+ minimum_holdout_latent_mse_improvement_fraction: 0.05
69
+ minimum_holdout_audio_ruler_improvement_fraction: 0.02
70
+ external_crop_sha256: 1c3d61f242412a78396665da14c8bcb4002dc93f5518fe81c0303fe9c0b77477
71
+ external_source_id: f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
72
+ publication:
73
+ private_dataset_root_mode: 448
74
+ private_dataset_file_mode: 384
75
+ evidence_root_mode: 493
76
+ evidence_file_mode: 420
77
+ manifest_written_last: true
78
+ atomic_directory_rename: true
configs/flow-inpaint-v1.yaml ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.masked-flow-inpaint-config.v1
2
+ capability: captured_condition_flow_latent_reconstruction_only
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ teacher:
7
+ prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
8
+ lyrics: "[instrumental]"
9
+ duration_seconds: 1.0
10
+ inference_steps: 12
11
+ guidance_scale: 1.7
12
+ train_seed_start: 1000
13
+ train_count: 64
14
+ validation_seed_start: 2000
15
+ validation_count: 16
16
+ heldout_seed_start: 3000
17
+ heldout_count: 16
18
+ adapter:
19
+ latent_channels: 128
20
+ frames: 86
21
+ condition_channels: 2048
22
+ flow_layers: 36
23
+ flow_hidden_size: 2048
24
+ lora_rank: 4
25
+ lora_scale: 1.0
26
+ region_embedding_count: 3
27
+ expected_trainable_parameters: 1775616
28
+ hole:
29
+ minimum_frames: 16
30
+ maximum_frames: 32
31
+ minimum_left_context_frames: 16
32
+ minimum_right_context_frames: 16
33
+ frame_samples: 512
34
+ training:
35
+ steps: 2000
36
+ batch_size: 8
37
+ seed: 731942
38
+ maximum_learning_rate: 0.0002
39
+ minimum_learning_rate: 0.00002
40
+ beta1: 0.9
41
+ beta2: 0.999
42
+ epsilon: 0.00000001
43
+ weight_decay: 0.0
44
+ gradient_clip_norm: 1.0
45
+ validation_interval_steps: 200
46
+ gradient_checkpointing: true
47
+ parameter_dtype: float32
48
+ compute_dtype: bfloat16
49
+ inference:
50
+ euler_steps: 30
51
+ guidance_scale: 1.7
52
+ evaluation_mask_seed: 880301
53
+ evaluation:
54
+ latent_std: 2.206683874130249
55
+ minimum_median_hole_latent_nmse_improvement_fraction: 0.05
56
+ minimum_median_hole_audio_ruler_improvement_fraction: 0.05
57
+ audio_ruler_config_path: flow-encoder-v1.yaml
58
+ publication:
59
+ private_dataset_root_mode: 448
60
+ private_dataset_file_mode: 384
61
+ evidence_root_mode: 493
62
+ evidence_file_mode: 420
63
+ manifest_written_last: true
64
+ atomic_directory_rename: true
65
+
66
+ # This pilot reconstructs masked regions of captured Music3 Flow state only.
67
+ # It does not accept arbitrary WAV, expose native RVQ tokens, or prove FIM/generalization.
configs/flow-prepend-v1.yaml ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.flow-prepend-config.v1
2
+ capability: local_one_second_captured_condition_renderer_latent_prefix_reconstruction_only
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ dataset:
7
+ schema_version: music3lab.flow-encoder-teachers.v2
8
+ semantic_digest: 0b8fb727a380b5a57373f12fb4ffd31eca65c2bdf855d4eee404c2380b3e5607
9
+ source_config_file_sha256: 3aaf6d48f755ce13ca031e2326c7caf830a53fba2054f827099317f4efe436b8
10
+ source_config_semantic_digest: de5f75e471772ac1aeb9805fce87c3293135f96fd21936e058cef8afa97f62c5
11
+ train_count: 64
12
+ validation_count: 16
13
+ heldout_count: 16
14
+ adapter:
15
+ latent_channels: 128
16
+ frames: 86
17
+ condition_channels: 2048
18
+ flow_layers: 36
19
+ flow_hidden_size: 2048
20
+ lora_rank: 4
21
+ lora_scale: 1.0
22
+ region_embedding_count: 4
23
+ expected_trainable_parameters: 1777664
24
+ masks:
25
+ minimum_frames: 16
26
+ maximum_frames: 32
27
+ minimum_left_context_frames: 16
28
+ minimum_right_context_frames: 16
29
+ two_sided_fraction: 0.5
30
+ prefix_fraction: 0.5
31
+ frame_samples: 512
32
+ training:
33
+ steps: 2000
34
+ batch_size: 8
35
+ seed: 731942
36
+ maximum_learning_rate: 0.0002
37
+ minimum_learning_rate: 0.00002
38
+ beta1: 0.9
39
+ beta2: 0.999
40
+ epsilon: 0.00000001
41
+ weight_decay: 0.0
42
+ gradient_clip_norm: 1.0
43
+ validation_interval_steps: 200
44
+ gradient_checkpointing: true
45
+ parameter_dtype: float32
46
+ compute_dtype: bfloat16
47
+ inference:
48
+ euler_steps: 30
49
+ guidance_scale: 1.7
50
+ evaluation_mask_seed: 880301
51
+ evaluation:
52
+ latent_std: 2.206683874130249
53
+ minimum_median_prefix_latent_nmse_improvement_fraction: 0.05
54
+ minimum_median_prefix_audio_ruler_improvement_fraction: 0.05
55
+ audio_ruler_config_path: flow-encoder-v1.yaml
56
+ publication:
57
+ private_dataset_root_mode: 448
58
+ private_dataset_file_mode: 384
59
+ evidence_root_mode: 493
60
+ evidence_file_mode: 420
61
+ manifest_written_last: true
62
+ atomic_directory_rename: true
63
+
64
+ # This reuses authenticated one-second captured Flow state and evaluates only a
65
+ # local renderer-latent prefix. It is not song-intro/native-AR prepend,
66
+ # arbitrary-WAV editing, FIM, native-token inference, or generalization.
configs/inversion-e3.yaml ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.inversion-e3-config.v1
2
+ experiment_label: external_audio_latent_inversion_no_oracle
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ external_source:
7
+ kind: opaque_user_audio
8
+ raw_file_sha256: f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
9
+ raw_file_size: 31437094
10
+ container: wav
11
+ codec: pcm_s32le
12
+ source_sample_rate: 48000
13
+ source_channels: 2
14
+ source_frames: 3929624
15
+ retain_original_bytes: true
16
+ persist_human_metadata: false
17
+ canonicalization:
18
+ engine: ffmpeg
19
+ executable: /usr/bin/ffmpeg
20
+ executable_sha256: 36d94a605d612e4090d1b8aec889d0c0801c6eafb1593c90f5c0dfd2e2966a45
21
+ version_line: ffmpeg version 4.4.2-0ubuntu0.22.04.1 Copyright (c) 2000-2021 the FFmpeg developers
22
+ input_transport: retained_source_bytes_via_stdin
23
+ operation_order: resample_full_source_then_trim_output_frames
24
+ filtergraph: aresample=44100,atrim=start_sample=1719900:end_sample=1763932,asetpts=PTS-STARTPTS
25
+ output_encoding: stereo_interleaved_little_endian_float32
26
+ output_sample_rate: 44100
27
+ output_channels: 2
28
+ output_frames: 44032
29
+ output_start_frame: 1719900
30
+ output_end_frame: 1763932
31
+ interleaved_f32le_sha256: 22d81730b21386f516bbb586f3310acb7aaed75b157a52dca9f0c7c76ef4b3bd
32
+ channel_major_tensor_sha256: b08e725241d7e3158e1a286ba9279e903b1a6a2f3f6c749af276ff02be0017b1
33
+ deterministic_float_wav_sha256: 2645ff4f920bb559cd8e4898f91e06c0f43bbfda715ccab2f08da1477440f3b7
34
+ execution:
35
+ device: cuda
36
+ required_device_name_substring: H100
37
+ deterministic_algorithms: true
38
+ restart_seeds: [101, 103, 107, 109]
39
+ master_latent_dtype: float32
40
+ fp32_decoder_dtype: float32
41
+ exact_decoder_dtype: bfloat16
42
+ trajectory_stride_steps: 1
43
+ latent_channels: 128
44
+ latent_frames: 86
45
+ latent_initialization_mean: 0.07592402398586273
46
+ latent_initialization_std: 2.206683874130249
47
+ initial_batch_sha256: d907a81862d25e1bf3bbf54b78b801708f723ba6b654759bbee813c304aec2f1
48
+ initial_restart_sha256:
49
+ - 6832687067d4c42e6b2f634c5e1648d12418259064c1fc71acb792e1c2ebefad
50
+ - eaa511076d9185d4f95dfc6f3aa6feae56f2002ecb8606c338dc1e12ce26e700
51
+ - 73ef18c7dda783bb61f0cfe4a4e0d71a5e17f1bfdd9455de6054e77228c7fdbb
52
+ - bcd53046e276188aee1bf4af4bfd8dddaf07513f67eaa7893158d055ce1dc4e5
53
+ optimizer:
54
+ name: adam
55
+ beta1: 0.9
56
+ beta2: 0.999
57
+ epsilon: 0.00000001
58
+ weight_decay: 0.0
59
+ gradient_clip_norm: 1.0
60
+ warmup_steps: 20
61
+ schedule: linear_warmup_then_cosine_inclusive
62
+ loss:
63
+ definition_version: fixed-bf16-ruler-v1
64
+ stft_center: false
65
+ stft_fft_sizes: [256, 512, 1024, 2048]
66
+ stft_hop_sizes: [64, 128, 256, 512]
67
+ legacy_mrstft_epsilon: 0.0000001
68
+ envelope_windows: [63, 255, 1023]
69
+ time_denominator_epsilon: 0.00000001
70
+ complex_stft_denominator_epsilon: 0.00000001
71
+ mid_side_floor_fraction: 0.0001
72
+ relative_envelope_denominator_epsilon: 0.00000001
73
+ fp32_audio_weights_start:
74
+ time_nmse: 1.0
75
+ complex_stft_nmse: 0.25
76
+ legacy_mrstft: 0.5
77
+ mid_side_nmse: 0.1
78
+ relative_envelope: 0.1
79
+ fixed_selection_weights:
80
+ time_nmse: 2.0
81
+ complex_stft_nmse: 1.0
82
+ legacy_mrstft: 0.05
83
+ mid_side_nmse: 0.5
84
+ relative_envelope: 0.02
85
+ fp32_prior_start: 0.02
86
+ fp32_prior_end: 0.005
87
+ bf16_prior_start: 0.005
88
+ bf16_prior_end: 0.001
89
+ distribution_prior:
90
+ definition: population_moment_and_tail_v1
91
+ latent_mean: 0.07592402398586273
92
+ latent_std: 2.206683874130249
93
+ variance_epsilon: 0.00000001
94
+ tail_threshold: 6.0
95
+ tail_weight: 0.1
96
+ formula: prior=mean(z)^2+(sqrt(mean((z-mean(z))^2)+1e-8)-1)^2+0.1*mean(relu(abs(z)-6)^2)
97
+ oracle_distance_used: false
98
+ selection:
99
+ score: fixed_exact_bf16_audio_ruler
100
+ evaluate_every_optimizer_step: true
101
+ per_restart_best: true
102
+ stage_transition_uses_fp32_stage_best: true
103
+ restart_tie_rule: earliest_stage_then_step
104
+ final_tie_rule: lowest_restart_index
105
+ evaluator_computed_after_lock: true
106
+ prior_excluded: true
107
+ experiment:
108
+ experiment_id: P2-E3
109
+ kind: external_audio_inversion
110
+ initialization: four_independent_seeded_prior_draws
111
+ fp32_stage:
112
+ steps: 800
113
+ maximum_learning_rate: 0.03
114
+ minimum_learning_rate: 0.003
115
+ bf16_stage:
116
+ steps: 1200
117
+ maximum_learning_rate: 0.015
118
+ minimum_learning_rate: 0.0015
119
+ audio_only_thresholds:
120
+ minimum_objective_improvement_fraction: 0.20
121
+ minimum_median_objective_improvement_fraction: 0.40
122
+ maximum_waveform_mae: 0.06
123
+ minimum_correlation: 0.50
124
+ minimum_si_sdr_db: -1.0
125
+ high_fidelity_minimum_si_sdr_db: 20.0
126
+ high_fidelity_minimum_unscaled_snr_db: 18.0
127
+ high_fidelity_maximum_loudness_error_db: 0.5
128
+ high_fidelity_maximum_stereo_correlation_error: 0.05
129
+ publication:
130
+ root_mode: 493
131
+ experiment_directory_mode: 448
132
+ file_mode: 420
133
+ manifest_written_last: true
134
+ atomic_directory_rename: true
135
+ notes:
136
+ capability: continuous Flow-VAE latent inversion from external waveform; not WAV-to-RVQ encoding
137
+ latent_quality_claim_allowed: false
138
+ human_metadata_persisted: false
139
+ quality_claim_policy: audio-only feasibility and high-fidelity tiers are separate
configs/inversion-v1.yaml ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.inversion-config.v1.2
2
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
3
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
4
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
5
+ oracle:
6
+ kind: short
7
+ run_id: 2a1bc3404f7a0c1c7bc03fc3813df5e8a3769aa9b413121f1669edd097353368
8
+ sampling_rate: 44100
9
+ channels: 2
10
+ samples: 44032
11
+ latent_channels: 128
12
+ latent_frames: 86
13
+ execution:
14
+ device: cuda
15
+ required_device_name_substring: H100
16
+ vocoder_dtype: bfloat16
17
+ master_latent_dtype: float32
18
+ deterministic_algorithms: true
19
+ restart_seeds: [101, 103, 107, 109]
20
+ trace_interval_steps: 10
21
+ loss:
22
+ charbonnier_epsilon: 0.001
23
+ stft_epsilon: 0.0000001
24
+ stft_center: false
25
+ definition_version: mrstft-center-false-unscaled-snr-v1
26
+ stft_fft_sizes: [256, 512, 1024, 2048]
27
+ stft_hop_sizes: [64, 128, 256, 512]
28
+ envelope_windows: [63, 255, 1023]
29
+ weights:
30
+ waveform_charbonnier: 1.0
31
+ mrstft: 0.15
32
+ mid_side_charbonnier: 0.25
33
+ multiscale_envelope: 0.15
34
+ latent_prior: 0.000001
35
+ experiments:
36
+ - experiment_id: P2-E1
37
+ kind: perturbed_oracle_recovery
38
+ steps: 240
39
+ learning_rate: 0.05
40
+ gradient_clip_norm: 2.0
41
+ initialization:
42
+ distribution: perturbed_oracle
43
+ mean: 0.0
44
+ std: 0.08
45
+ thresholds:
46
+ minimum_objective_improvement_fraction: 0.75
47
+ minimum_median_objective_improvement_fraction: 0.40
48
+ maximum_waveform_mae: 0.008
49
+ minimum_correlation: 0.99
50
+ minimum_si_sdr_db: 22.0
51
+ maximum_latent_rmse: 0.08
52
+ high_fidelity_minimum_si_sdr_db: 20.0
53
+ high_fidelity_minimum_unscaled_snr_db: 18.0
54
+ high_fidelity_maximum_loudness_error_db: 0.5
55
+ high_fidelity_maximum_stereo_correlation_error: 0.05
56
+ - experiment_id: P2-E2
57
+ kind: generated_waveform_inversion
58
+ steps: 240
59
+ learning_rate: 0.08
60
+ gradient_clip_norm: 2.0
61
+ initialization:
62
+ distribution: random_normal
63
+ mean: 0.075924
64
+ std: 2.206684
65
+ thresholds:
66
+ minimum_objective_improvement_fraction: 0.20
67
+ minimum_median_objective_improvement_fraction: 0.40
68
+ maximum_waveform_mae: 0.06
69
+ minimum_correlation: 0.50
70
+ minimum_si_sdr_db: -1.0
71
+ maximum_latent_rmse: null
72
+ high_fidelity_minimum_si_sdr_db: 20.0
73
+ high_fidelity_minimum_unscaled_snr_db: 18.0
74
+ high_fidelity_maximum_loudness_error_db: 0.5
75
+ high_fidelity_maximum_stereo_correlation_error: 0.05
76
+ selection:
77
+ criterion: final_optimization_objective
78
+ evaluator_metrics_computed_after_selection: true
79
+ permit_evaluator_metric_selection: false
80
+ tie_rule: lowest_restart_index
81
+ publication:
82
+ root_mode: 493
83
+ experiment_directory_mode: 448
84
+ file_mode: 420
85
+ manifest_written_last: true
86
+ atomic_directory_rename: true
87
+ notes:
88
+ target_provenance: >-
89
+ Exact BF16-generated Phase-0 short_parity waveform and final continuous
90
+ Flow-VAE latent oracle; this experiment does not encode a WAV into native
91
+ RVQ tokens.
92
+ quality_claim_policy: >-
93
+ Report no inversion-quality success unless every preregistered threshold
94
+ for that experiment passes.
configs/inversion-v2.yaml ADDED
@@ -0,0 +1,175 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.inversion-v2-config.v2
2
+ experiment_label: latent_only_continuation_not_optimizer_resume
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ authority_migration:
7
+ kind: v1.1_to_v1.3_authority_only
8
+ preregistered_config_sha256: dba79de7191ed4a4242f4fecb691314b04fde945a1eb9f4e1fa1c97a6a017a3f
9
+ preserved_sections_sha256: 287b6aed4d19ec63f3f78bd1ee414eb3a6c1f56b7f6dc961fb5e0ea4c02ff7a1
10
+ only_changed_top_level_keys: [schema_version, v1_authority, authority_migration]
11
+ v1_authority:
12
+ schema_version: music3lab.inversion-session.v1.3
13
+ root_name: inversion-v1.3-3632094
14
+ session_file_sha256: bf615c719ddaa53fd300512867694447ea91c6703fa709f250b28484ecf68445
15
+ session_semantic_digest: 1554159fc7a5e364de063e3c8345c3cf05426288aa23bd0039957026aabc3a7c
16
+ config_file_sha256: e79362ec680b6e2af163c17da8942675eb31d4084aa301abd143646cf0d51779
17
+ config_semantic_digest: 09ee76c6aeec8a91733391b386f1c8f043df5a9919beb9acb29b718199e2900c
18
+ project_git_commit: 36320944c624faefba78459ae07f9e0d1faea874
19
+ project_source_sha256: 9b2a741ed6ee5ff9763a2fb81aee0194d37d469c8e676c213ddf5aa9418a1921
20
+ adapter_semantic_digest: 2c7cd7a868d1bd3f0ea670c44bd9a1b806b58fbb4b2293264b04a279e7d285ea
21
+ oracle_semantic_digest: b1f4ed241802c3d52c1a8fef8ca882177f2f3342d07e586e50ffa0f095609347
22
+ p2e1_manifest_file_sha256: 18503958f437f3c7aef5a807bac1f26d7822acf55f5c5c8813ddfcc8d2a82987
23
+ p2e1_manifest_semantic_digest: 2dc5af49efa39a313519ea908a6a4694c98663b780c79f99caf8272271d5e161
24
+ p2e2_manifest_file_sha256: b27a1093a53673ba8f10eb40f6afbc7ae1ad43728037ea306a2ba59a7a7e1760
25
+ p2e2_manifest_semantic_digest: ed9fa2c63aa6c8c8f41b0fb3adab474f9ce1ab8cb803d62d945f8db4e2249b6e
26
+ p2e1_latents_file_sha256: dd56e055c39d7344d501daa9ae47c4c4da07c583278751bf4b1ccc4e747f0e57
27
+ p2e2_latents_file_sha256: 8e4e974adb69e00911bb041c957398db7dd07317fe6120550d31929781d2a571
28
+ p2e1_final_latents_sha256: e687c11b2eb406821e8b972d1edcd5a956af587b9787230fcda089d23689d336
29
+ p2e1_final_restart_sha256:
30
+ - 6f0276e1d6139570ba8789bdb84a7463cdbcdebcc030473e1e314693b658bb73
31
+ - 5db91d5d300a9d3bd5aa171113cf7b6849075bb00a0be23dc279d19e7fb28c53
32
+ - 996cb44f54f01c5a0495bb4390b7a5322023f52eba28b3f20729bb669dba62d3
33
+ - 92fa103dee0a71a787e9c5037d935104032e665e783e231886c45a6b6ea1dc83
34
+ p2e2_final_latents_sha256: 6287ffa20f336188e2e24c9bdd0f992518127c7eefdb05123adcf62e36d100c6
35
+ p2e2_final_restart_sha256:
36
+ - 518f00fa4b22a2a334efca0efafd669f6d9e39b092a84934bdf440753ca680b7
37
+ - 2a25806f5065c4f89e796f135e8162af60b1e781209c923bbc6f134deac020aa
38
+ - e0a59a93c125502fa104ecd5efc2a619c10fc0c5a8efc9d65991ab747659edac
39
+ - 144aafa55d311a8d05eb01566d61d7b223d13ca8d06347760ee0b237751782c3
40
+ p2e1_seeded_restart_sha256:
41
+ - bc8b5700a710bad531dd41cea022a3e74d317c174027e6efa5144acc02c443fe
42
+ - 52f2d55b630cb86ae26cf6f7ef656c73b1e0bf2fbec4023e2f3e91cfba5e3d9d
43
+ - 6b5ce53cdbcf27f594c5e5a68434af4f6f3d3dca6bfa1db2a9e75f083d32b831
44
+ - cae9150a6e73da5c0cec5196cb3390cb78ca956ebef1cbcacfa5dc069788cd51
45
+ oracle:
46
+ kind: short
47
+ run_id: 2a1bc3404f7a0c1c7bc03fc3813df5e8a3769aa9b413121f1669edd097353368
48
+ sampling_rate: 44100
49
+ channels: 2
50
+ samples: 44032
51
+ latent_channels: 128
52
+ latent_frames: 86
53
+ execution:
54
+ device: cuda
55
+ required_device_name_substring: H100
56
+ deterministic_algorithms: true
57
+ restart_seeds: [101, 103, 107, 109]
58
+ master_latent_dtype: float32
59
+ fp32_decoder_dtype: float32
60
+ exact_decoder_dtype: bfloat16
61
+ trajectory_stride_steps: 1
62
+ optimizer:
63
+ name: adam
64
+ beta1: 0.9
65
+ beta2: 0.999
66
+ epsilon: 0.00000001
67
+ weight_decay: 0.0
68
+ gradient_clip_norm: 1.0
69
+ warmup_steps: 20
70
+ schedule: linear_warmup_then_cosine_inclusive
71
+ loss:
72
+ definition_version: fixed-bf16-ruler-v1
73
+ stft_center: false
74
+ stft_fft_sizes: [256, 512, 1024, 2048]
75
+ stft_hop_sizes: [64, 128, 256, 512]
76
+ legacy_mrstft_epsilon: 0.0000001
77
+ envelope_windows: [63, 255, 1023]
78
+ time_denominator_epsilon: 0.00000001
79
+ complex_stft_denominator_epsilon: 0.00000001
80
+ mid_side_floor_fraction: 0.0001
81
+ relative_envelope_denominator_epsilon: 0.00000001
82
+ fp32_audio_weights_start:
83
+ time_nmse: 1.0
84
+ complex_stft_nmse: 0.25
85
+ legacy_mrstft: 0.5
86
+ mid_side_nmse: 0.1
87
+ relative_envelope: 0.1
88
+ fixed_selection_weights:
89
+ time_nmse: 2.0
90
+ complex_stft_nmse: 1.0
91
+ legacy_mrstft: 0.05
92
+ mid_side_nmse: 0.5
93
+ relative_envelope: 0.02
94
+ fp32_prior_start: 0.02
95
+ fp32_prior_end: 0.005
96
+ bf16_prior_start: 0.005
97
+ bf16_prior_end: 0.001
98
+ distribution_prior:
99
+ definition: population_moment_and_tail_v1
100
+ latent_mean: 0.07592402398586273
101
+ latent_std: 2.206683874130249
102
+ variance_epsilon: 0.00000001
103
+ tail_threshold: 6.0
104
+ tail_weight: 0.1
105
+ formula: prior=mean(z)^2+(sqrt(mean((z-mean(z))^2)+1e-8)-1)^2+0.1*mean(relu(abs(z)-6)^2)
106
+ oracle_distance_used: false
107
+ selection:
108
+ score: fixed_exact_bf16_audio_ruler
109
+ evaluate_every_optimizer_step: true
110
+ per_restart_best: true
111
+ stage_transition_uses_fp32_stage_best: true
112
+ restart_tie_rule: earliest_stage_then_step
113
+ final_tie_rule: lowest_restart_index
114
+ evaluator_computed_after_lock: true
115
+ prior_excluded: true
116
+ engineering_progress:
117
+ label: engineering_progress_3db_v1
118
+ definition: 10*log10(v1_fixed_bf16_score/v2_fixed_bf16_score)
119
+ minimum_db: 3.0
120
+ affects_quality_claim: false
121
+ experiments:
122
+ - experiment_id: P2-E1
123
+ kind: perturbed_oracle_recovery
124
+ initialization: original_seeded_perturbations_not_v1_optimizer_state
125
+ fp32_stage:
126
+ steps: 600
127
+ maximum_learning_rate: 0.02
128
+ minimum_learning_rate: 0.002
129
+ bf16_stage:
130
+ steps: 600
131
+ maximum_learning_rate: 0.008
132
+ minimum_learning_rate: 0.0008
133
+ thresholds:
134
+ minimum_objective_improvement_fraction: 0.75
135
+ minimum_median_objective_improvement_fraction: 0.40
136
+ maximum_waveform_mae: 0.008
137
+ minimum_correlation: 0.99
138
+ minimum_si_sdr_db: 22.0
139
+ maximum_latent_rmse: 0.08
140
+ high_fidelity_minimum_si_sdr_db: 20.0
141
+ high_fidelity_minimum_unscaled_snr_db: 18.0
142
+ high_fidelity_maximum_loudness_error_db: 0.5
143
+ high_fidelity_maximum_stereo_correlation_error: 0.05
144
+ - experiment_id: P2-E2
145
+ kind: generated_waveform_inversion
146
+ initialization: all_four_v1_final_latents_not_optimizer_resume
147
+ fp32_stage:
148
+ steps: 800
149
+ maximum_learning_rate: 0.03
150
+ minimum_learning_rate: 0.003
151
+ bf16_stage:
152
+ steps: 1200
153
+ maximum_learning_rate: 0.015
154
+ minimum_learning_rate: 0.0015
155
+ thresholds:
156
+ minimum_objective_improvement_fraction: 0.20
157
+ minimum_median_objective_improvement_fraction: 0.40
158
+ maximum_waveform_mae: 0.06
159
+ minimum_correlation: 0.50
160
+ minimum_si_sdr_db: -1.0
161
+ maximum_latent_rmse: null
162
+ high_fidelity_minimum_si_sdr_db: 20.0
163
+ high_fidelity_minimum_unscaled_snr_db: 18.0
164
+ high_fidelity_maximum_loudness_error_db: 0.5
165
+ high_fidelity_maximum_stereo_correlation_error: 0.05
166
+ publication:
167
+ root_mode: 493
168
+ experiment_directory_mode: 448
169
+ file_mode: 420
170
+ manifest_written_last: true
171
+ atomic_directory_rename: true
172
+ notes:
173
+ capability: continuous Flow-VAE latent-only continuation; not WAV-to-RVQ encoding
174
+ optimizer_resume: false
175
+ quality_claim_policy: preserve v1.1 feasibility/high-fidelity gates; engineering progress is separately labeled
configs/learned-audio-continuation-v1.yaml ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.learned-audio-continuation-config.v1
2
+ capability: "continuous-latent local continuation; non-native-token; not long-song"
3
+ sample_rate: 44100
4
+ samples_per_window: 44032
5
+ target_parameterization: target_minus_repeat_tail
6
+ generated_latent_parameterization: repeat_tail_plus_generated_residual
7
+ corpus:
8
+ manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
9
+ splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
10
+ semantic_digest: 6cd403e9cd25e61d062af6caa9c6b136b751c7f668f1af7d0972989963ac3452
11
+ train_count: 542
12
+ validation_count: 68
13
+ heldout_count: 68
14
+ guard_samples: 220500
15
+ crop_namespace: music3lab.learned-continuation.crop.v1
16
+ source_exclusive: true
17
+ one_pair_per_source: true
18
+ projector:
19
+ checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
20
+ source_gate: REJECTED_teacher_latent_nmse_regression
21
+ champion_eligible: false
22
+ frozen: true
23
+ latent_shape: [128, 86]
24
+ condition_shape: [86, 2048]
25
+ condition_repeat: 16
26
+ terminal_zero_trim: exact_only
27
+ target_visible_to_conditioner: false
28
+ adapter:
29
+ flow_layers: 36
30
+ hidden_size: 2048
31
+ rank: 4
32
+ scale: 1.0
33
+ expected_trainable_parameters: 1769472
34
+ base_flow_frozen: true
35
+ vocoder_frozen: true
36
+ training:
37
+ steps: 1200
38
+ batch_size: 16
39
+ seed: 260815
40
+ maximum_learning_rate: 0.00005
41
+ minimum_learning_rate: 0.000005
42
+ beta1: 0.9
43
+ beta2: 0.999
44
+ epsilon: 0.00000001
45
+ weight_decay: 0.0
46
+ gradient_clip_norm: 1.0
47
+ validation_interval_steps: 100
48
+ gradient_checkpointing: true
49
+ compute_dtype: bfloat16
50
+ parameter_dtype: float32
51
+ inference:
52
+ euler_steps: 30
53
+ guidance_scale: 1.7
54
+ noise_seed: 73021
55
+ overlap_samples: 1024
56
+ evaluation:
57
+ same_noise_across_conditions: true
58
+ condition_baselines: [zero_context, unrelated_context]
59
+ direct_baselines: [repeat_tail, roll_tail]
60
+ strict_median_latent_nmse_beats_every_baseline: true
61
+ strict_median_audio_ruler_beats_every_baseline: true
62
+ benchmark_batch_sizes: [4, 8, 16]
63
+ seam:
64
+ metric_domain: composed_output
65
+ derivative_absolute_floor: 0.00001
66
+ rms_absolute_floor: 0.0001
67
+ maximum_median_boundary_derivative_ratio: 2.0
68
+ maximum_median_overlap_rms_log_error: 1.5
69
+ claims:
70
+ native_tokens: false
71
+ text_conditioning: false
72
+ captured_condition: false
73
+ future_audio_conditioning: false
74
+ arbitrary_wav_local_continuation: true
75
+ long_song_generalization: false
76
+ specialist_promoted: false
77
+ handcrafted_baseline_reclassified: false
configs/long-reference-style-v1.yaml ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.long-reference-style-config.v1
2
+ capability: long_reference_internal_text_bridge_plus_postrender_ranking
3
+ model_id: MiniMaxAI/MiniMax-Music3
4
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ sample_rate: 44100
7
+ channels: 2
8
+ frame_rate: 25
9
+ chunk_frames: 200
10
+ chunk_hop: 100
11
+ inference_steps: 12
12
+ guidance_scale: 1.7
13
+ profile_samples: 44032
14
+ profile_maximum_crops: 8
15
+ specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
16
+ specialist_role: REJECTED_EXTERNAL_SPECIALIST_STYLE_SCORER_ONLY
17
+ policies:
18
+ fast: {candidates: 2, duration_seconds: 30.0}
19
+ balanced: {candidates: 4, duration_seconds: 60.0}
20
+ maximum: {candidates: 8, duration_seconds: 90.0}
21
+ independent_ar_flow_seed_supported: false
22
+ direct_model_audio_conditioning: false
23
+ native_negative_prompt_used: false
24
+ full_semantic_style_transfer: false
25
+ artistic_quality_claim: false
26
+ vocal_absence_guaranteed: false
configs/native-state-distill.yaml ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.native-state-distill-config.v1
2
+ capability: captured_music3_multiclip_only
3
+ source:
4
+ teacher_manifest_sha256: 6ceffbe5aab6ea808f66ada57bbe3e91244f96db15f5038040269afbc47b3071
5
+ primary_splits_only: true
6
+ diagnostic_clips_excluded: true
7
+ teacher:
8
+ semantics: official_post_cfg_conditional_top50_semantic_mask_sampling_distribution
9
+ storage: finite_normalized_float32_log_probabilities
10
+ temperature: 1.0
11
+ audio_code_offset: 151675
12
+ semantic_vocab_size: 16384
13
+ cfg_scale: 1.5
14
+ conditional_top_k: 50
15
+ sampling_top_k: 50
16
+ initialization:
17
+ description: selected multiclip checkpoint before distillation
18
+ checkpoint_sha256: ed8505a00438a3d3b1a605584313f6214dee37e13dba6f07aab4a5c42d700c43
19
+ training:
20
+ seed: 929292
21
+ steps: 5000
22
+ batch_size: 8
23
+ validation_interval_steps: 250
24
+ maximum_learning_rate: 0.0003
25
+ minimum_learning_rate: 0.00003
26
+ gradient_clip_norm: 10.0
27
+ weight_decay: 0.0
28
+ optimization_scope: semantic_head_only_to_preserve_native_state_heads
29
+ loss_weights:
30
+ c0_kl: 5.0
31
+ feedback: 2.0
32
+ global_hidden: 0.5
33
+ residual: 0.0
34
+ selection:
35
+ dataset: primary_unseen_prompt_validation_only
36
+ objective: minimum_c0_distillation_kl
37
+ state_nmse_regression_tolerance: 0.000001
38
+ terminal_gate:
39
+ heldout_hard_c0_ce_maximum: 9.604060527839234
40
+ heldout_kl_improvement_fraction_minimum: 0.10
41
+ heldout_feedback_nmse_maximum: 1.114844
42
+ heldout_global_nmse_maximum: 0.844784
43
+ rounding_tolerance: 0.000001
44
+ native_tokenizer: false
45
+ arbitrary_external_music: false
46
+ generalization_claim: false
configs/native-state-multiclip.yaml ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.native-state-multiclip-config.v1
2
+ capability: captured_music3_multiclip_only
3
+ model:
4
+ width: 96
5
+ heads: 4
6
+ conformer_layers: 8
7
+ mel_bins: 80
8
+ conv_kernel: 15
9
+ residual_heads: disabled
10
+ capture:
11
+ duration_seconds: 1.0
12
+ inference_steps: 12
13
+ lyrics: "[instrumental]"
14
+ fresh_prompts:
15
+ train_a: "Instrumental UK garage, 132 BPM, swung drums, deep sub bass, bright chopped chords."
16
+ train_b: "Instrumental ambient electronic, 90 BPM, evolving pads, soft pulse, spacious granular texture."
17
+ validation_c: "Instrumental Latin electronic, 110 BPM, syncopated hand percussion, warm bass, vivid mallet motif."
18
+ heldout_d: "Instrumental acoustic folk, 76 BPM, fingerpicked guitar, brushed percussion, intimate strings."
19
+ training:
20
+ seed: 919191
21
+ steps: 10000
22
+ batch_size: 8
23
+ validation_interval_steps: 500
24
+ maximum_learning_rate: 0.0003
25
+ minimum_learning_rate: 0.00003
26
+ gradient_clip_norm: 10.0
27
+ weight_decay: 0.0
28
+ loss_weights:
29
+ c0: 5.0
30
+ feedback: 2.0
31
+ global_hidden: 0.5
32
+ residual: 0.0
33
+ selection:
34
+ dataset: primary_unseen_prompt_validation_only
35
+ objective: stage0_weighted_total
36
+ scale_up_gate:
37
+ heldout_feedback_nmse_improvement_minimum: 0.05
38
+ heldout_global_nmse_improvement_minimum: 0.05
39
+ heldout_c0_ce_maximum_margin_below_uniform: 0.1
40
+ train_vs_heldout_error_ratio_maximum: 2.0
41
+ error_ratio_definition: heldout_error_divided_by_train_error
42
+ native_tokenizer: false
43
+ arbitrary_external_music: false
44
+ generalization_claim: false
configs/native-state-stage0.yaml ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.native-state-stage0.v1
2
+ capability: one_clip_music3_native_state_posterior_alignment_proof
3
+ teacher:
4
+ source_case: existing_native_token_adapter_train_seed_1000
5
+ prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
6
+ lyrics: "[instrumental]"
7
+ seed: 1000
8
+ duration_seconds: 1.0
9
+ inference_steps: 12
10
+ first_sampled_row_is_priming: true
11
+ emitted_frame_count: 25
12
+ model:
13
+ width: 96
14
+ heads: 4
15
+ conformer_layers: 8
16
+ mel_bins: 80
17
+ conv_kernel: 15
18
+ residual_heads: disabled
19
+ loss:
20
+ c0: 5.0
21
+ feedback: 2.0
22
+ global_hidden: 0.5
23
+ residual: 0.0
24
+ alignment:
25
+ minimum_offset: -8
26
+ maximum_offset: 8
27
+ calibration_steps: 80
28
+ calibration_repeats: 2
29
+ full_overfit_steps: 6000
30
+ learning_rate: 0.001
31
+ acceptance:
32
+ c0_exact_frames: 25
33
+ feedback_normalized_mse_maximum: 0.001
34
+ feedback_cosine_minimum: 0.999
35
+ global_hidden_normalized_mse_maximum: 0.001
36
+ global_hidden_cosine_minimum: 0.999
37
+ native_tokenizer: false
38
+ arbitrary_external_music: false
39
+ generalization_claim: false
configs/native-state-stage1-local.yaml ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.native-stage1-local-config.v1
2
+ capability: captured_music3_only
3
+ corpus:
4
+ train: 800
5
+ validation: 112
6
+ heldout: 112
7
+ frames: 25
8
+ lyrics: "[instrumental]"
9
+ inference_steps: 12
10
+ augmentation:
11
+ checkpoint_every_clips: 256
12
+ generation_calls: 0
13
+ training:
14
+ seed: 969696
15
+ steps: 5000
16
+ batch_size: 8
17
+ validation_interval_steps: 250
18
+ maximum_learning_rate: 0.0003
19
+ minimum_learning_rate: 0.00003
20
+ gradient_clip_norm: 10.0
21
+ weight_decay: 0.0
22
+ hidden_nmse_weight: 1.0
23
+ hidden_cosine_weight: 1.0
24
+ guided_kl_weight: 0.25
25
+ hard_ce_weight: 1.0
26
+ selection: lowest_validation_mean_residual_ce_subject_to_finite_and_stage0_identity
27
+ bootstrap:
28
+ seed: 979797
29
+ resamples: 2000
30
+ confidence: 0.95
31
+ native_tokenizer: false
32
+ arbitrary_external_music: false
configs/native-state-stage1.yaml ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.native-stage1-config.v1
2
+ capability: captured_music3_only
3
+ corpus:
4
+ train: 800
5
+ validation: 112
6
+ heldout: 112
7
+ train_prompt_ids: 25
8
+ validation_prompt_ids: 4
9
+ heldout_prompt_ids: 4
10
+ duration_seconds: 1.0
11
+ inference_steps: 12
12
+ lyrics: "[instrumental]"
13
+ shard_size: 32
14
+ training:
15
+ seed: 939393
16
+ steps: 5000
17
+ batch_size: 8
18
+ validation_interval_steps: 250
19
+ maximum_learning_rate: 0.0003
20
+ minimum_learning_rate: 0.00003
21
+ gradient_clip_norm: 10.0
22
+ weight_decay: 0.0
23
+ scheduled_sampling_teacher_fraction: 0.4
24
+ hard_ce_weight: 1.0
25
+ soft_kl_weight: 0.25
26
+ selection: lowest_validation_mean_residual_hard_ce_subject_to_finite_and_exact_stage0
27
+ bootstrap:
28
+ seed: 949494
29
+ resamples: 2000
30
+ confidence: 0.95
31
+ terminal:
32
+ each_depth_ce_improvement: 0.03
33
+ mean_ce_improvement: 0.05
34
+ top1_minimum_depths: 5
35
+ top1_improvement_pp: 1.0
36
+ autoregressive_vs_h_only_improvement: 0.02
37
+ full_row_paired_bootstrap_lower_ci_strictly_above: 0.0
38
+ native_tokenizer: false
39
+ arbitrary_external_music: false
40
+ generalization_claim: false
configs/native-token-adapter-v1.yaml ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.native-token-adapter-config.v1
2
+ capability: captured_music3_waveform_to_rvq_row_prediction_only
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ flow_encoder_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
7
+ teacher:
8
+ prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
9
+ lyrics: "[instrumental]"
10
+ duration_seconds: 1.0
11
+ inference_steps: 12
12
+ guidance_scale: 1.7
13
+ train_seed_start: 1000
14
+ train_count: 64
15
+ validation_seed_start: 2000
16
+ validation_count: 16
17
+ heldout_seed_start: 3000
18
+ heldout_count: 16
19
+ training:
20
+ steps: 2000
21
+ batch_size: 8
22
+ seed: 515151
23
+ maximum_learning_rate: 0.0003
24
+ minimum_learning_rate: 0.00003
25
+ weight_decay: 0.0001
26
+ gradient_clip_norm: 1.0
27
+ validation_interval_steps: 100
28
+ evaluation:
29
+ predicted_render_seed: 9091
30
+ top_k: 5
31
+ native_tokenizer: false
32
+ arbitrary_external_music: false
33
+ generalization_claim: false
configs/objective-only-v1.yaml ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.objective-only-config.v1
2
+ corpus:
3
+ corpus_label: opaque-user-audio-objective-v1
4
+ source_count: 8
5
+ raw_file_mode: 292
6
+ crop_file_mode: 292
7
+ directory_mode: 493
8
+ manifest_file_mode: 420
9
+ preserve_human_metadata: false
10
+ source_identity: raw_sha256
11
+ split:
12
+ algorithm: prior_exposure_then_lexicographic_sha256_v1
13
+ prior_exposure_source_ids:
14
+ - f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
15
+ calibration_count: 1
16
+ validation_count: 3
17
+ heldout_count: 4
18
+ heldout_sealed: true
19
+ additions_require_new_schema: true
20
+ crop:
21
+ algorithm: sha256_domain_uint64_mod_interior_v1
22
+ domain: "music3lab.objective-only.crop.v1"
23
+ output_sample_rate: 44100
24
+ output_channels: 2
25
+ output_frames: 44032
26
+ guard_frames: 220500
27
+ operation_order: resample_full_source_then_trim_output_frames
28
+ canonicalizer:
29
+ executable: /usr/bin/ffmpeg
30
+ executable_sha256: 36d94a605d612e4090d1b8aec889d0c0801c6eafb1593c90f5c0dfd2e2966a45
31
+ version_line: ffmpeg version 4.4.2-0ubuntu0.22.04.1 Copyright (c) 2000-2021 the FFmpeg developers
32
+ input_transport: retained_source_bytes_via_stdin
33
+ output_encoding: stereo_interleaved_little_endian_float32
34
+ sources:
35
+ - source_id: 054f93d3933c4f3e4839af1d901b71f6ba0d80b5dc6abd80776a22545fca13a9
36
+ raw_size: 2208136
37
+ container: mp3
38
+ codec: mp3
39
+ sample_rate: 44100
40
+ channels: 2
41
+ time_base: 1/14112000
42
+ duration_ticks: 1298349864
43
+ canonical_frames: 4057344
44
+ canonical_full_sha256: 8587aabd20d0d7709b31b1e8178519fe9f4596e37bb9e2e3a2e73e1b90cfba80
45
+ split: validation
46
+ crop_start_frame: 784654
47
+ crop_interleaved_sha256: afb4e733fa35ce4651776f4fa8fcb3bf37c80e65d5ddbd3a90c804a860d6e5a9
48
+ - source_id: 5c6ea28a706eacea653309ab40548ca324f7dd1f5cc1707a16d4ef93e46ec310
49
+ raw_size: 2344078
50
+ container: m4a
51
+ codec: aac
52
+ sample_rate: 48000
53
+ channels: 2
54
+ time_base: 1/48000
55
+ duration_ticks: 6847488
56
+ canonical_frames: 6289190
57
+ canonical_full_sha256: 939a017c0c4554f5b072f6dd6ff00a73c0142478168db2c4f7bc7ceef0d60eaf
58
+ split: validation
59
+ crop_start_frame: 4591657
60
+ crop_interleaved_sha256: 457d2ce0dfdb35963d8835d194943c28f4c99146637a8299a98180a90cdb7888
61
+ - source_id: 994f9a6b74ebc3ed9bd3d25433b77c0415c9f53a8c73441f2891b46211321e14
62
+ raw_size: 33375668
63
+ container: wav
64
+ codec: pcm_s16le
65
+ sample_rate: 48000
66
+ channels: 2
67
+ time_base: 1/48000
68
+ duration_ticks: 8343871
69
+ canonical_frames: 7665932
70
+ canonical_full_sha256: 49cb8dcaa9183bbca46db61a63d5ef61d56e2e5bf170853c00a407074138f2f0
71
+ split: validation
72
+ crop_start_frame: 2163717
73
+ crop_interleaved_sha256: 8f0ea8da159164cc03e231e39e15ce7259e62ad03236cfb96a6ead91a05db6e7
74
+ - source_id: 9b3291618c7f00a988fa1093243c0c575fd593d9c1d952db3b44a2326424116d
75
+ raw_size: 33180184
76
+ container: wav
77
+ codec: pcm_s16le
78
+ sample_rate: 48000
79
+ channels: 2
80
+ time_base: 1/48000
81
+ duration_ticks: 8295000
82
+ canonical_frames: 7621032
83
+ canonical_full_sha256: 9e7b8a4eadd86be966ce3bc9d799eddbffda4e6720ac6f6bffb06527b5645f45
84
+ split: heldout_test
85
+ crop_start_frame: 6372962
86
+ crop_interleaved_sha256: 07cbed859a775f9e9296066386dedf068916022e488140891ab2d4ea561d451f
87
+ - source_id: c2368b62f03fb46455d429524e3613d2ba8746d1e4c3435fef73e62369b5be9d
88
+ raw_size: 2284542
89
+ container: m4a
90
+ codec: aac
91
+ sample_rate: 48000
92
+ channels: 2
93
+ time_base: 1/48000
94
+ duration_ticks: 6648832
95
+ canonical_frames: 6106674
96
+ canonical_full_sha256: 9cf7287782a1548a6fbc98cd0aebb689c1cd6675c404455bc75052ce8787e23b
97
+ split: heldout_test
98
+ crop_start_frame: 232371
99
+ crop_interleaved_sha256: c63f87185e7e1337efef296aa5da0a725ee694264647bcde6292673fd7abb692
100
+ - source_id: e182bd207c5a02276cc04f0b932ac47842ab61855b33d762bcfa344042b75c91
101
+ raw_size: 31204096
102
+ container: wav
103
+ codec: pcm_s16le
104
+ sample_rate: 48000
105
+ channels: 2
106
+ time_base: 1/48000
107
+ duration_ticks: 7800000
108
+ canonical_frames: 7166250
109
+ canonical_full_sha256: eeece5e958b727c483396d93df29a51909f6edfe234885410888f82f5b9530e6
110
+ split: heldout_test
111
+ crop_start_frame: 4023033
112
+ crop_interleaved_sha256: 84804551646a84f958d30e027575fcdd9001757bd06dae13b2f7ede5480c79f7
113
+ - source_id: f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
114
+ raw_size: 31437094
115
+ container: wav
116
+ codec: pcm_s32le
117
+ sample_rate: 48000
118
+ channels: 2
119
+ time_base: 1/48000
120
+ duration_ticks: 3929624
121
+ canonical_frames: 3610343
122
+ canonical_full_sha256: 392112ff198a7f2fa5ffb92f6083a21e2c7532b826fb5afcb86acf6d72f38d8c
123
+ split: calibration
124
+ crop_start_frame: 1103007
125
+ crop_interleaved_sha256: d2afcf0f0f034c32ad0d044cfd7cb89dde634aa97d2c7efa1f81f82b9969cc52
126
+ - source_id: ff777c257a9f8044bbe08eb2933d6f4073a22770cf04ae527b8fe768982e6fea
127
+ raw_size: 13153696
128
+ container: wav
129
+ codec: pcm_s16le
130
+ sample_rate: 44100
131
+ channels: 2
132
+ time_base: 1/44100
133
+ duration_ticks: 3288354
134
+ canonical_frames: 3288354
135
+ canonical_full_sha256: da3cd2960f30d0c58845ce3b6ca1fe807b5c9a237d34a9edf13ccd96b0fea3a6
136
+ split: heldout_test
137
+ crop_start_frame: 1360904
138
+ crop_interleaved_sha256: d85d4d805e99e305b571ad2a464e7bd1a80cfffb02249129e5441aebdeda7a23
139
+ metric_policy:
140
+ mandatory_boards: [signal, reference, boundary, diversity]
141
+ signal:
142
+ minimum_rms: 0.0001
143
+ maximum_peak: 1.5
144
+ maximum_clipped_fraction: 0.05
145
+ reference:
146
+ maximum_median_waveform_mae: 0.06
147
+ minimum_median_correlation: 0.5
148
+ minimum_median_si_sdr_db: -1.0
149
+ minimum_median_unscaled_snr_db: -1.0
150
+ boundary:
151
+ maximum_edge_derivative: 0.5
152
+ maximum_edge_rms_ratio: 4.0
153
+ diversity:
154
+ minimum_unique_fraction: 1.0
155
+ minimum_pairwise_feature_distance: 0.0001
156
+ optional_evaluators:
157
+ clap_alignment: false
158
+ frechet_audio_distance: false
159
+ learned_music_quality: false
160
+ candidate_policy:
161
+ selection_allowed_splits: [calibration, validation]
162
+ heldout_split: heldout_test
163
+ selection_must_be_locked: true
164
+ missing_mandatory_result: INSUFFICIENT_EVIDENCE
165
+ publication:
166
+ root_mode: 493
167
+ leaf_mode: 448
168
+ file_mode: 420
169
+ manifest_written_last: true
170
+ atomic_directory_rename: true
configs/prefix-seed-search-v1.yaml ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.prefix-seed-search-config.v1
2
+ capability: least_measurable_splice_discontinuity_among_four_captured_prefix_candidates
3
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
4
+ expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ source:
7
+ prepend_project_commit: 82795b470538ec1c08793a559016f1f9fcd38ab8
8
+ prepend_config_file_sha256: 995f216302b682e495aaf3a5a7499ca1d595f1a31e7c7c43ddee7558fb1d6c48
9
+ prepend_config_semantic_digest: ee16251758454669bc58710752ce976f15739d4f242fb7a3f5d7e206ab46b554
10
+ prepend_bundle_semantic_digest: 181832f93cb0a41a12c07fdf54ddf817ac4f20fa1586e01f7956b8e0da45bdec
11
+ prepend_metrics_semantic_digest: 121386375685bf599c406ed434808b42af4bf6f3101d7bfcc19554bba3395f7b
12
+ prepend_metrics_file_sha256: 6d9926f9c44634f9fabb4afe51e274d39cb0d1a6e2f4f7ae0285d935eb5ad0f3
13
+ prepend_adapter_file_sha256: 0048d3d727f0faeea53fcc136f777917e58435de8cba8a35b293cae6ad5aab28
14
+ prepend_adapter_state_sha256: a60c74c557035a19ebb4138b683015e4f4358b9881f28cdb775a16c2fc52ba1a
15
+ teacher_v2_semantic_digest: 0b8fb727a380b5a57373f12fb4ffd31eca65c2bdf855d4eee404c2380b3e5607
16
+ heldout_index: 0
17
+ heldout_seed: 3000
18
+ search:
19
+ seeds: [101, 103, 107, 109]
20
+ sequential_single_model_load: true
21
+ prefix_mask_seed: 880301
22
+ expected_prefix_frames: 22
23
+ expected_latent_frames: 86
24
+ frame_samples: 512
25
+ expected_audio_samples: 44032
26
+ channels: 2
27
+ sampling_rate: 44100
28
+ euler_steps: 30
29
+ guidance_scale: 1.7
30
+ gates:
31
+ minimum_rms: 0.0001
32
+ maximum_abs_peak: 1.5
33
+ clipped_abs_threshold: 1.0
34
+ maximum_clipped_fraction: 0.05
35
+ require_finite: true
36
+ require_expected_length: true
37
+ require_known_suffix_exact: true
38
+ require_unique_audio_sha256: true
39
+ selection:
40
+ join_derivative_half_window_samples: 512
41
+ edge_rms_window_samples: 1024
42
+ edge_rms_epsilon: 0.00000001
43
+ lexicographic_fields:
44
+ - join_max_abs_derivative
45
+ - join_edge_rms_ratio
46
+ - seed
47
+ publication:
48
+ evidence_root_mode: 493
49
+ evidence_file_mode: 420
50
+ manifest_written_last: true
51
+ atomic_directory_rename: true
52
+ artistic_quality_claim: false
53
+ stock_win_claim: false
54
+
55
+ # The selected result is only the least measurable splice discontinuity among
56
+ # these four fixed candidates. It is not an artistic-quality or stock-win claim.
configs/prompt-free-v1.yaml ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.prompt-free-config.v1
2
+ capability: prompt_free_internal_text_compatibility_bridge
3
+ model_id: MiniMaxAI/MiniMax-Music3
4
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ sample_rate: 44100
7
+ channels: 2
8
+ inference_steps: 12
9
+ guidance_scale: 1.7
10
+ minimum_duration_seconds: 1.0
11
+ maximum_duration_seconds: 30.0
12
+ lyrics_literal: "[instrumental]"
13
+ source_prior_count: 32
14
+ text_origin: internal_generated
15
+ null_conditioning: false
16
+ artistic_quality_claim: false
17
+ rendered_diversity_claim: false
18
+ guaranteed_no_vocals: false
19
+ arbitrary_audio_reference_conditioning: false
configs/promptfree-bestofn-v1.yaml ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.promptfree-bestofn-config.v1
2
+ capability: prompt_free_long_form_best_of_n_internal_text_bridge
3
+ model_id: MiniMaxAI/MiniMax-Music3
4
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ sample_rate: 44100
7
+ channels: 2
8
+ frame_rate: 25
9
+ chunk_frames: 200
10
+ chunk_hop: 100
11
+ inference_steps: 12
12
+ guidance_scale: 1.7
13
+ source_prior_count: 32
14
+ duration_minimum_seconds: 1
15
+ duration_maximum_seconds: 360
16
+ diversity_epsilon: 1.0e-6
17
+ policies:
18
+ fast:
19
+ candidates: 2
20
+ default_duration_seconds: 30.0
21
+ balanced:
22
+ candidates: 4
23
+ default_duration_seconds: 60.0
24
+ maximum:
25
+ candidates: 8
26
+ default_duration_seconds: 90.0
27
+ internal_plan_only: true
28
+ independent_ar_flow_seed_supported: false
29
+ artistic_quality_claim: false
configs/reference-guided-append-v1.yaml ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.reference-guided-append-config.v1
2
+ capability: reference_guided_append
3
+ sample_rate: 44100
4
+ channels: 2
5
+ candidate_frames: 352768
6
+ crossfade_frames: 2048
7
+ seam_weight: 0.15
8
+ silence_rms: 0.0001
9
+ clipped_fraction_limit: 0.05
10
+ source_copy_correlation: 0.75
11
+ model_generation: false
12
+ native_history_conditioning: false
13
+ semantic_style_transfer: false
14
+ native_negative_prompt: false
configs/reference-ranked-pool-v1.yaml ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.reference-ranked-pool-config.v1
2
+ capability: reference_ranked_long_form_pool_not_direct_audio_conditioning
3
+ source_audio: /home/ubuntu/minimax-user-audio/incoming/loveonme_x_osh.wav
4
+ source_audio_sha256: ff777c257a9f8044bbe08eb2933d6f4073a22770cf04ae527b8fe768982e6fea
5
+ pool_root: /home/ubuntu/minimax-promptfree-bestofn-evidence/maximum90-c19c0c5
6
+ retained_candidate_ids: ["002", "003", "007"]
7
+ retained_candidate_wav_sha256:
8
+ "002": 72855ef9cf1e77e5aa718e956e27cec7a23b8e55b9ecb1b53229011a48409ddd
9
+ "003": 29e2c8afc5dd00c39bd8a79c278829d4ea8450703e9690bc963626ddea832739
10
+ "007": d5cbd5c920745e106b14aaf6659d95255f8b0f8d73a8b9f97ab47ecc00f72207
11
+ specialist_checkpoint: /home/ubuntu/minimax-external-finetune-evidence/interim678-78437a8/encoder.safetensors
12
+ specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
13
+ flow_config: /home/ubuntu/minimax-music3-release/configs/flow-encoder-v1.yaml
14
+ flow_config_sha256: 4a2474a50dd1e9e93c15cb5cf792e344531809a868b8f1b7405846a6012b47df
15
+ seed: 17090
16
+ negative: no clipping, no source copy, no tempo drift, no vocals
17
+ sample_rate: 44100
18
+ channels: 2
19
+ duration_seconds: 90.0
20
+ duration_tolerance_samples: 2048
21
+ profile_samples: 44032
22
+ profile_crops: 8
23
+ device: cpu
24
+ generate: false
25
+ train: false
26
+ output: /home/ubuntu/minimax-reference-ranked-pool-evidence/loveonme-exact90-f040ab1/selected.reference-ranked.wav
27
+ ffmpeg: /usr/bin/ffmpeg
configs/reference-style-v1.yaml ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.reference-style-config.v1
2
+ capability: reference_style_direct_latent_scorer_internal_text_bridge
3
+ model_id: MiniMaxAI/MiniMax-Music3
4
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ sample_rate: 44100
7
+ channels: 2
8
+ profile_samples: 44032
9
+ profile_maximum_crops: 8
10
+ default_duration_seconds: 8
11
+ default_candidates: 8
12
+ maximum_candidates: 8
13
+ inference_steps: 12
14
+ specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
15
+ specialist_role: REJECTED_EXTERNAL_SPECIALIST_STYLE_SCORER_ONLY
16
+ native_negative_prompt_used: false
17
+ direct_model_audio_conditioning: false
18
+ full_semantic_style_transfer: false
19
+ artistic_quality_claim: false
20
+ vocal_absence_guaranteed: false
configs/sample-bridge-v1.yaml ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.sample-bridge-config.v1
2
+ capability: mir_conditioned_internal_text_bridge_not_direct_audio_embedding
3
+ model_id: MiniMaxAI/MiniMax-Music3
4
+ model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
5
+ diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
6
+ sample_rate: 44100
7
+ channels: 2
8
+ inference_steps: 12
9
+ minimum_duration_seconds: 1.0
10
+ maximum_duration_seconds: 30.0
11
+ maximum_candidates: 8
12
+ clipping_fraction_limit: 0.05
13
+ silence_rms_limit: 0.0001
14
+ tempo_drift_limit_octaves: 0.20
15
+ near_copy_similarity_limit: 0.995
16
+ expected_frame_tolerance: 2048
17
+ lyrics_literal: "[instrumental]"
18
+ native_negative_prompt_used: false
19
+ direct_audio_embedding: false
20
+ full_style_match: false
21
+ native_token_continuation: false
22
+ artistic_quality_claim: false
23
+ vocal_absence_guaranteed: false
configs/waveform-causal-continuation-v1.yaml ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.waveform-causal-continuation-v1-config.v1
2
+ capability: arbitrary_waveform_causal_short_continuation_not_native_tokens_not_semantic_style
3
+ seed: 260816
4
+ steps: 12000
5
+ validation_interval: 500
6
+ sample_rate: 44100
7
+ history_samples: 65536
8
+ prediction_samples: 8192
9
+ source_endpoint_guard_samples: 220500
10
+ model_channels: [48, 64, 96, 128, 160]
11
+ bottleneck_dilations: [1, 2, 4, 8, 16, 32, 64, 128]
12
+ residual_scale: 0.5
13
+ benchmark_batch_sizes_h100: [24, 32]
14
+ batch_24gb: 4
15
+ accumulation_24gb: 8
16
+ max_lr: 0.0002
17
+ min_lr: 0.00002
18
+ weight_decay: 0.01
19
+ gradient_clip_norm: 1.0
20
+ validation_namespace: music3lab.waveform-causal-continuation-v1.validation.v1
21
+ heldout_namespace: music3lab.waveform-causal-continuation-v1.heldout.v1
22
+ manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
23
+ splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
24
+ loss_weights:
25
+ waveform_l1: 1.0
26
+ waveform_l2: 0.5
27
+ mrstft: 0.25
28
+ complex_stft_nmse: 0.15
29
+ mid_side: 0.10
30
+ boundary_first1024: 0.20
31
+ roll_tail_samples: 1024
32
+ nonsilent_rms_floor: 0.0001
33
+ noncopy_correlation_threshold: 0.995
34
+ seam:
35
+ derivative_absolute_floor: 0.00001
36
+ rms_absolute_floor: 0.0001
37
+ maximum_median_derivative_ratio: 2.0
38
+ maximum_median_join_rms_log_error: 1.5
39
+ augmentations:
40
+ gain_db: [-3.0, 3.0]
41
+ stereo_balance_db: [-1.5, 1.5]
42
+ shared_polarity_probability: 0.5
43
+ channel_swap_probability: 0.5
44
+ forbidden: [side_shift, noise, resample]
configs/waveform-right-context-prepend-v1.yaml ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: music3lab.waveform-right-context-prepend-v1-config.v1
2
+ capability: arbitrary_waveform_right_context_short_precursor_not_native_tokens_not_semantic_prepend
3
+ seed: 260816
4
+ steps: 12000
5
+ validation_interval: 500
6
+ sample_rate: 44100
7
+ prefix_samples: 8192
8
+ right_context_samples: 65536
9
+ source_endpoint_guard_samples: 220500
10
+ crop_alignment_samples: 512
11
+ stft: {n_fft: 1024, hop_length: 256, center: false, frames: 253}
12
+ encoder_channels: [48, 64, 96, 128, 160, 192]
13
+ transformer: {blocks: 6, width: 256, heads: 8, bidirectional: true}
14
+ target_queries: 29
15
+ benchmark_batch_sizes_h100: [24, 32]
16
+ max_lr: 0.0002
17
+ min_lr: 0.00002
18
+ weight_decay: 0.01
19
+ gradient_clip_norm: 1.0
20
+ validation_namespace: music3lab.waveform-right-context-prepend-v1.validation.v1
21
+ heldout_namespace: music3lab.waveform-right-context-prepend-v1.heldout.v1
22
+ manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
23
+ splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
24
+ loss_weights:
25
+ waveform_l1: 1.0
26
+ waveform_l2: 0.25
27
+ mrstft: 0.30
28
+ complex_stft_nmse: 0.20
29
+ mid_side: 0.10
30
+ seam_waveform_first_difference_last1024: 0.25
31
+ seam_rms_log: 0.10
32
+ nonsilent_rms_floor: 0.0001
33
+ user22_policy: ood_only_after_heldout_decision
34
+ augmentations:
35
+ gain_db: [-3.0, 3.0]
36
+ stereo_balance_db: [-1.5, 1.5]
37
+ shared_polarity_probability: 0.5
38
+ channel_swap_probability: 0.5
39
+ forbidden: [side_shift, noise, resample]
data/laion_disco/UPSTREAM_README.md ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ dataset_info:
4
+ features:
5
+ - name: song_id
6
+ dtype: string
7
+ - name: title
8
+ dtype: string
9
+ - name: artist_names
10
+ sequence: string
11
+ - name: artist_ids
12
+ sequence: string
13
+ - name: album_name
14
+ dtype: string
15
+ - name: album_id
16
+ dtype: string
17
+ - name: isExplicit
18
+ dtype: bool
19
+ - name: views
20
+ dtype: string
21
+ - name: duration
22
+ dtype: int64
23
+ splits:
24
+ - name: train
25
+ num_bytes: 2069255857
26
+ num_examples: 12320916
27
+ download_size: 750206954
28
+ dataset_size: 2069255857
29
+ configs:
30
+ - config_name: default
31
+ data_files:
32
+ - split: train
33
+ path: data/train-*
34
+ tags:
35
+ - music
36
+ pretty_name: LAION DISCO
37
+ size_categories:
38
+ - 10M<n<100M
39
+ ---
40
+
41
+
42
+ The LAION-DISCO-12M dataset contains 12M links to music on YouTube, inspired by the methodology of DISCO-10M. It contains song metadata (song_id, title, artist_names, artist_ids, album_name, album_id, isExplicit, views, duration) and YouTube URL, pointing to the original song on the public web. It does not contain any original audio samples and is thus an index dataset.
43
+
44
+ Starting from an initial seed list of artists, we can discover new artists by recursively exploring the artists listed in the "Fans might also like" section.
45
+ We explore the related artists graph for as long as we are able to find new artists.
46
+ For a given artist, we can extract their metadata, such as their name and number of subscribers, as well as a list of all of their songs and music videos.
47
+ Importantly, each song or music video is associated with a YouTube URL (obtained from its ID). The collected metadata fields are: song_id, title, artist_names, artist_ids, album_name, album_id, isExplicit, views, duration.
48
+
49
+ The authors of DISCO-10M used a seed list of 18 artists, chosen to represent a variety of genres. However, we found that this is not sufficient for exploring the artist graph of YouTube Music. Starting from this seed list, we were able to discover only 90,007 artists and 5,399,389 songs.
50
+
51
+ We therefore compiled a larger seed list by considering the artists that appear on YouTube Music charts of top songs by country and genre playlists.
52
+ This resulted in an initial list of 45,218 artists. The artist graph exploration starting from this seed list resulted in 250,516 artists and 12,648,485 songs.
53
+
54
+ This work was inspired by [DISCO-10M](https://arxiv.org/abs/2306.13512), consider citing them if you use this dataset.
data/laion_disco/candidates.summary.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "candidate_manifest_sha256": "533d8623268b18f87de2774ecf45656520f37edd8d5bb30c69f9bb6401231a07",
3
+ "counts": {
4
+ "album_cap_rejected": 110,
5
+ "artist_cap_rejected": 163,
6
+ "duration_filtered": 1179263,
7
+ "metadata_rows": 12320916,
8
+ "rows_scanned": 12320916
9
+ },
10
+ "dataset": "laion/LAION-DISCO-12M",
11
+ "first_rank": "0000004322543062a3737fdb182a7ead063119640afd3745ffe658991041948a",
12
+ "last_rank": "0077fb6c207e2a35049f7547f4ec40da5dbc1a497341391903475a347b37eb8a",
13
+ "revision": "6e7bf3758a77301e46a715af894fefd79bb1da53",
14
+ "schema_version": 1,
15
+ "selection": {
16
+ "album_cap": 1,
17
+ "artist_cap": 2,
18
+ "candidate_count": 20000,
19
+ "cap_scope": "all listed artist IDs; normalized names when IDs absent",
20
+ "intended_success_count": 2000,
21
+ "metadata_duration_seconds_inclusive": [
22
+ 45,
23
+ 360
24
+ ],
25
+ "pool_size": 250000,
26
+ "rank": "ascending sha256(revision + NUL + song_id)"
27
+ }
28
+ }