code and training scripts
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +5 -35
- .gitignore +43 -0
- DATA.md +60 -0
- FINDINGS.md +245 -0
- LICENSE +201 -0
- MODEL_CARD.md +59 -0
- NOTICE +26 -0
- README.md +145 -0
- REPORT_AUDIO_PREPEND.md +14 -0
- REPORT_REFERENCE_RANKED_POOL.md +30 -0
- REPORT_REFERENCE_STYLE.md +32 -0
- REPRODUCING.md +90 -0
- capture_generated_tokens.py +460 -0
- checkpoints/README.md +35 -0
- configs/acoustic-fim-v2.yaml +41 -0
- configs/audio-continuation-v1.yaml +25 -0
- configs/audio-fim-v1.yaml +18 -0
- configs/audio-prepend-v1.yaml +18 -0
- configs/autonomous-interim678-v1.yaml +20 -0
- configs/baseline.yaml +52 -0
- configs/captured-state-style-continuation-v1.yaml +36 -0
- configs/continuation-tier-a.yaml +25 -0
- configs/external-flow-encoder-interim-678-v1.yaml +63 -0
- configs/external-flow-encoder-v1.yaml +56 -0
- configs/flow-encoder-v1.yaml +78 -0
- configs/flow-inpaint-v1.yaml +67 -0
- configs/flow-prepend-v1.yaml +66 -0
- configs/inversion-e3.yaml +139 -0
- configs/inversion-v1.yaml +94 -0
- configs/inversion-v2.yaml +175 -0
- configs/learned-audio-continuation-v1.yaml +77 -0
- configs/long-reference-style-v1.yaml +26 -0
- configs/native-state-distill.yaml +46 -0
- configs/native-state-multiclip.yaml +44 -0
- configs/native-state-stage0.yaml +39 -0
- configs/native-state-stage1-local.yaml +32 -0
- configs/native-state-stage1.yaml +40 -0
- configs/native-token-adapter-v1.yaml +33 -0
- configs/objective-only-v1.yaml +170 -0
- configs/prefix-seed-search-v1.yaml +56 -0
- configs/prompt-free-v1.yaml +19 -0
- configs/promptfree-bestofn-v1.yaml +29 -0
- configs/reference-guided-append-v1.yaml +14 -0
- configs/reference-ranked-pool-v1.yaml +27 -0
- configs/reference-style-v1.yaml +20 -0
- configs/sample-bridge-v1.yaml +23 -0
- configs/waveform-causal-continuation-v1.yaml +44 -0
- configs/waveform-right-context-prepend-v1.yaml +39 -0
- data/laion_disco/UPSTREAM_README.md +54 -0
- data/laion_disco/candidates.summary.json +28 -0
.gitattributes
CHANGED
|
@@ -1,35 +1,5 @@
|
|
| 1 |
-
*
|
| 2 |
-
*.
|
| 3 |
-
*.
|
| 4 |
-
*.
|
| 5 |
-
*.
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
-
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
-
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
| 1 |
+
* text=auto eol=lf
|
| 2 |
+
*.wav binary
|
| 3 |
+
*.safetensors binary
|
| 4 |
+
*.png binary
|
| 5 |
+
*.jpg binary
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
.gitignore
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# --- Python ---
|
| 2 |
+
.venv/
|
| 3 |
+
venv/
|
| 4 |
+
__pycache__/
|
| 5 |
+
*.py[cod]
|
| 6 |
+
.pytest_cache/
|
| 7 |
+
*.egg-info/
|
| 8 |
+
build/
|
| 9 |
+
dist/
|
| 10 |
+
|
| 11 |
+
# --- models / weights / audio (never commit) ---
|
| 12 |
+
models/
|
| 13 |
+
artifacts/
|
| 14 |
+
checkpoints/**/*.safetensors
|
| 15 |
+
*.safetensors
|
| 16 |
+
*.pt
|
| 17 |
+
*.pth
|
| 18 |
+
*.ckpt
|
| 19 |
+
*.bin
|
| 20 |
+
*.wav
|
| 21 |
+
*.mp3
|
| 22 |
+
*.m4a
|
| 23 |
+
*.flac
|
| 24 |
+
*.npy
|
| 25 |
+
*.npz
|
| 26 |
+
|
| 27 |
+
# --- downloaded dataset audio (keep metadata, never the audio) ---
|
| 28 |
+
data/**/files/
|
| 29 |
+
data/**/canonical/
|
| 30 |
+
data/**/*.wav
|
| 31 |
+
|
| 32 |
+
# --- secrets / credentials (never commit) ---
|
| 33 |
+
.env
|
| 34 |
+
*.pem
|
| 35 |
+
*.key
|
| 36 |
+
id_rsa*
|
| 37 |
+
id_ed25519*
|
| 38 |
+
*hf_token*
|
| 39 |
+
**/known_user22_* # private author-song exclusion hashes/filenames
|
| 40 |
+
|
| 41 |
+
# --- os cruft ---
|
| 42 |
+
.DS_Store
|
| 43 |
+
Thumbs.db
|
DATA.md
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Data
|
| 2 |
+
|
| 3 |
+
Music3Lab ships **dataset metadata only**. No audio of any kind is included or
|
| 4 |
+
redistributed.
|
| 5 |
+
|
| 6 |
+
## LAION-DISCO-12M (external real-music corpus)
|
| 7 |
+
|
| 8 |
+
[LAION-DISCO-12M](https://huggingface.co/datasets/laion/LAION-DISCO-12M) is an
|
| 9 |
+
**index** dataset: it contains YouTube video IDs and tags, licensed **Apache-2.0**
|
| 10 |
+
for the metadata, and explicitly contains **no audio samples**. Audio referenced
|
| 11 |
+
by those IDs is *not* part of the dataset and is *not* redistributed here.
|
| 12 |
+
|
| 13 |
+
### What is shipped in this repo (`data/laion_disco/`)
|
| 14 |
+
|
| 15 |
+
- `interim_tranche_678/manifest.jsonl` — the 678 accepted tracks as records of:
|
| 16 |
+
YouTube `source_id`/`source_url`, LAION `source_dataset_revision`, canonical
|
| 17 |
+
audio content hashes (`canonical_sha256`, `pcm_sha256`), sample rate / channels
|
| 18 |
+
/ frame counts, and split assignment. **No audio.**
|
| 19 |
+
- `interim_tranche_678/splits.json` — deterministic, source-level train/val/heldout
|
| 20 |
+
split (algorithm, seed, assignment, counts).
|
| 21 |
+
- `interim_tranche_678/summary.json`, `candidates.summary.json`,
|
| 22 |
+
`tranche.stopped.json`, `environment.json` — provenance and stop-reason records.
|
| 23 |
+
- `UPSTREAM_README.md` — the corpus builder's own notes.
|
| 24 |
+
|
| 25 |
+
### What is NOT shipped
|
| 26 |
+
|
| 27 |
+
- ❌ Any downloaded audio (`files/*.wav`, `canonical/`).
|
| 28 |
+
- ❌ The full 20k candidate list is included only as a summary; the raw
|
| 29 |
+
`candidates.jsonl` and the private author-song exclusion hashes
|
| 30 |
+
(`known_user22_*`) are **not** published.
|
| 31 |
+
|
| 32 |
+
### Reproducing the corpus
|
| 33 |
+
|
| 34 |
+
`scripts/data/laion_ingest.py` deterministically selects candidates from the
|
| 35 |
+
pinned LAION revision (`6e7bf3758a77301e46a715af894fefd79bb1da53`), then uses
|
| 36 |
+
`yt-dlp` + `ffmpeg` to fetch and canonicalize audio **on your own machine**.
|
| 37 |
+
`scripts/data/laion_freeze_interim.py` freezes an immutable, hash-verified
|
| 38 |
+
tranche and split. Both scripts contain instance-specific absolute paths from the
|
| 39 |
+
original run and need their `ROOT`/paths adjusted before use.
|
| 40 |
+
|
| 41 |
+
> ⚠️ **You are responsible** for complying with YouTube's Terms of Service and
|
| 42 |
+
> applicable copyright when downloading any audio. Music3Lab neither hosts nor
|
| 43 |
+
> distributes that audio. The original run stopped at 678/2,000 tracks when
|
| 44 |
+
> YouTube returned bot-verification challenges — expect the same and throttle
|
| 45 |
+
> accordingly.
|
| 46 |
+
|
| 47 |
+
## The author's own songs (private)
|
| 48 |
+
|
| 49 |
+
The project also evaluated on the author's own tracks (8, then 22 after
|
| 50 |
+
deduplication). These are **private by default** and are **not** in this repo,
|
| 51 |
+
not in the LAION splits, and were **never** used for training or checkpoint
|
| 52 |
+
selection — only as out-of-distribution evaluation. Only their content **hashes**
|
| 53 |
+
were used, to guarantee they never leaked into training. If you want a public
|
| 54 |
+
demo, publish only short excerpts of audio you own and are licensed to share.
|
| 55 |
+
|
| 56 |
+
## MiniMax-Music3 teacher data
|
| 57 |
+
|
| 58 |
+
Some experiments use WAV/latent/token pairs *captured from* MiniMax-Music3
|
| 59 |
+
generations. These captures are derivatives of the base model and are **not**
|
| 60 |
+
shipped. `scripts/` can regenerate them from the model you download yourself.
|
FINDINGS.md
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# MiniMax Music 3 `dav.pth` investigation
|
| 2 |
+
|
| 3 |
+
## Conclusion
|
| 4 |
+
|
| 5 |
+
The released `dav.pth` does **not** contain the native Music 3 waveform-to-RVQ tokenizer. It contains a continuous
|
| 6 |
+
Flow-VAE analysis/synthesis checkpoint: a waveform convolutional `encoder`, Gaussian posterior heads (`mean_proj` and
|
| 7 |
+
`logs_proj`), continuous `flow` transforms, and a waveform `decoder`. There is no RVQ/VQ quantizer and there are no
|
| 8 |
+
acoustic codebook embeddings. The `encoder.*` name is therefore real but is not the discrete encoder required by the
|
| 9 |
+
proposed continuation path.
|
| 10 |
+
|
| 11 |
+
The safe CPU-only inspection of release revision `fbdf52fbaaca799592917417eb05f1899f1255ec` found:
|
| 12 |
+
|
| 13 |
+
| Prefix | Tensor count | Representative evidence |
|
| 14 |
+
|---|---:|---|
|
| 15 |
+
| `encoder.*` | 119 | `encoder.block.0.weight_v [64,1,7]`; final trunk `encoder.block.6.weight_v [1024,1024,3]` |
|
| 16 |
+
| `mean_proj.*` | 2 | `mean_proj.weight [64,1024,1]` |
|
| 17 |
+
| `logs_proj.*` | 2 | `logs_proj.weight [64,1024,1]` |
|
| 18 |
+
| `dec_in_proj.*` | 2 | `dec_in_proj.weight [1024,64,1]` |
|
| 19 |
+
| `decoder.*` | 119 | first convolution `[1536,1024,7]`; final convolution `[1,96,7]` |
|
| 20 |
+
| `flow.*` | 304 | `flow.flows.0.pre.weight [256,32,1]` through `flow.flows.6.post.weight [32,256,1]` |
|
| 21 |
+
|
| 22 |
+
That is 548 tensors, all FP32, containing 122,904,034 values (491,616,136 tensor bytes). The file is 491,817,450
|
| 23 |
+
bytes and has SHA-256 `52adde6c6c52cca872f549449cd7b608677d1e69f4a98eee22da3403e63b56e8`. It is a flat
|
| 24 |
+
`OrderedDict`: there is no `generator` key, no `62000_generator` key or wrapper, and zero names matching
|
| 25 |
+
`quantizer`, `vq`, `rvq`, `codebook`, `codec`, or `tokenizer`. It also contains no serialized config or architecture
|
| 26 |
+
metadata. Tensor shapes plus released source are sufficient to instantiate the decoder, but cannot instantiate a
|
| 27 |
+
quantizer that has neither parameters nor configuration.
|
| 28 |
+
|
| 29 |
+
The reported community `62000_generator` wrapper is not present in this exact released file. `inspect_dav.py` still
|
| 30 |
+
recursively supports `generator`, `62000_generator`, and `state_dict`-style wrappers so a different checkpoint can be
|
| 31 |
+
checked without assuming it has the release layout. Names, shapes, and generic config can flag a candidate weight set,
|
| 32 |
+
but the inspector deliberately keeps `native_discrete_tokenizer_complete` and `can_encode_native_music3_tokens` false
|
| 33 |
+
until an exact compatible executable architecture/API has been implemented and verified.
|
| 34 |
+
|
| 35 |
+
## The actual token and rendering paths
|
| 36 |
+
|
| 37 |
+
Music 3 uses eight discrete streams at 25 frames/s. They are language-model symbols used to produce continuous
|
| 38 |
+
conditioning hidden states; they are **not** DAV latent codes and are never passed directly to the DAV decoder.
|
| 39 |
+
|
| 40 |
+
```text
|
| 41 |
+
caption + lyrics + <|audio_start|>
|
| 42 |
+
|
|
| 43 |
+
v
|
| 44 |
+
Global Qwen LM: c0 in [0, 16383] (vocabulary token c0 + 151675)
|
| 45 |
+
|
|
| 46 |
+
v
|
| 47 |
+
Local RVQ-depth LM: c1..c7, each in [0, 1023]
|
| 48 |
+
|
|
| 49 |
+
+--> frame token row [c0,c1,...,c7]
|
| 50 |
+
|
|
| 51 |
+
+--> global hidden + 7 depth hiddens = 8 * 4096
|
| 52 |
+
frame_hiddens [1, frames, 32768]
|
| 53 |
+
|
|
| 54 |
+
v
|
| 55 |
+
condition projection + flow-matching DiT
|
| 56 |
+
|
|
| 57 |
+
v
|
| 58 |
+
continuous Flow-VAE latent [B,128,T]
|
| 59 |
+
|
|
| 60 |
+
v
|
| 61 |
+
DAV dec_in_proj + decoder -> waveform
|
| 62 |
+
```
|
| 63 |
+
|
| 64 |
+
The token trajectory convention used by the provided comparator is `[frames, 8]`; `[8, frames]` is accepted only
|
| 65 |
+
when explicitly declared as `codebooks_first`. No heuristic transposition is performed.
|
| 66 |
+
|
| 67 |
+
The per-frame autoregressive details are:
|
| 68 |
+
|
| 69 |
+
1. The global LM samples only `<|audio_end|>` or the 16,384 c0 vocabulary IDs beginning at offset 151675.
|
| 70 |
+
2. The local depth model begins with the projected global hidden and the projected c0 embedding, then samples c1
|
| 71 |
+
through c7 autoregressively. It has seven 1,024-way output heads.
|
| 72 |
+
3. For feedback to the global LM, c0 uses the global token embedding at `c0 + 151675`. Each residual code uses the
|
| 73 |
+
Qwen checkpoint's `model.audio_extra_embedding.weight` in seven disjoint 1,024-row bands. Those seven embeddings
|
| 74 |
+
are summed with c0 and scaled by `1/sqrt(8)`. This is an **LM embedding table**, not the absent waveform
|
| 75 |
+
quantizer's acoustic centroids.
|
| 76 |
+
4. Codebooks are frame-aligned (one row contains all eight codes); there is no delayed-codebook layout in this path.
|
| 77 |
+
Conditional and classifier-free-unconditional rows are paired during sampling. The first decode immediately after
|
| 78 |
+
`<|audio_start|>` primes the feedback loop and is not emitted as an acoustic hidden frame. Each subsequent emitted
|
| 79 |
+
frame concatenates one 4096-wide global hidden with seven 4096-wide local hiddens.
|
| 80 |
+
5. The 25 Hz rate is explicit in Diffusers and also follows `24000 / 960` in the condition encoder configuration.
|
| 81 |
+
|
| 82 |
+
Special IDs are fixed by the music tokenizer:
|
| 83 |
+
|
| 84 |
+
| Token | ID |
|
| 85 |
+
|---|---:|
|
| 86 |
+
| `<|im_start|>` | 151644 |
|
| 87 |
+
| `<|im_end|>` | 151645 |
|
| 88 |
+
| `<|audio_cfg|>` | 151654 |
|
| 89 |
+
| `<|audio_start|>` | 151669 |
|
| 90 |
+
| `<|audio_end|>` | 151670 |
|
| 91 |
+
| `<|caption_start|>` | 151671 |
|
| 92 |
+
| `<|caption_end|>` | 151672 |
|
| 93 |
+
| `<|lyrics_start|>` | 151673 |
|
| 94 |
+
| `<|lyrics_end|>` | 151674 |
|
| 95 |
+
| first c0/audio-code vocabulary ID | 151675 |
|
| 96 |
+
|
| 97 |
+
The Qwen shards contain the global audio-symbol embeddings plus the local autoregressive depth decoder
|
| 98 |
+
(`model.audio_extra_embedding.*` and `model.audio_decoder.*`). They contain no waveform encoder, nearest-neighbor
|
| 99 |
+
quantizer, RVQ codebooks, or equivalent audio-tokenizer module. A model that can *predict* token IDs from text is not
|
| 100 |
+
therefore able to infer those IDs from a waveform.
|
| 101 |
+
|
| 102 |
+
## Successful official H100 generation capture
|
| 103 |
+
|
| 104 |
+
A pinned end-to-end Diffusers run succeeded on the remote NVIDIA H100 using model revision
|
| 105 |
+
`fbdf52fbaaca799592917417eb05f1899f1255ec`, Diffusers revision
|
| 106 |
+
`90b4e34e79a86ec5e7f2437634fe95ecd2108796`, CUDA with bfloat16 weights, and seed 7. The exact request used prompt
|
| 107 |
+
`Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass.`, lyrics
|
| 108 |
+
`[instrumental]`, requested duration 1.0 seconds, and 30 inference steps. It made 26 calls to the official
|
| 109 |
+
`_generate_depth_codes`: one priming call followed by 25 emitted frames. After skipping only that priming row,
|
| 110 |
+
`music3_internal_tokens.npy` is a genuine internal `[25,8]` trajectory at 25 Hz.
|
| 111 |
+
|
| 112 |
+
The observed inclusive per-codebook ranges were:
|
| 113 |
+
|
| 114 |
+
| Codebook | Minimum | Maximum | Legal range |
|
| 115 |
+
|---|---:|---:|---:|
|
| 116 |
+
| c0 | 1012 | 16163 | 0..16383 |
|
| 117 |
+
| c1 | 95 | 984 | 0..1023 |
|
| 118 |
+
| c2 | 25 | 1005 | 0..1023 |
|
| 119 |
+
| c3 | 42 | 950 | 0..1023 |
|
| 120 |
+
| c4 | 83 | 941 | 0..1023 |
|
| 121 |
+
| c5 | 2 | 967 | 0..1023 |
|
| 122 |
+
| c6 | 3 | 984 | 0..1023 |
|
| 123 |
+
| c7 | 67 | 957 | 0..1023 |
|
| 124 |
+
|
| 125 |
+
The corresponding official renderer output, `music3_generated_1s.wav`, is 44.1 kHz stereo with 44,032 samples per
|
| 126 |
+
channel (0.9984580499 seconds). The reusable `capture_generated_tokens.py` utility reproduces this observation by
|
| 127 |
+
temporarily monkeypatching the official depth-code function, validating its paired `[2,8]` result, restoring the
|
| 128 |
+
function after generation, and saving only calls after the priming row. Before model loading it resolves the requested
|
| 129 |
+
Hub revision to a concrete snapshot commit, verifies the exact Diffusers commit, rejects dirty tracked files in a
|
| 130 |
+
source checkout, and hashes both the relevant Diffusers source file and the capture script. It writes temporary WAV,
|
| 131 |
+
NPY, and JSON files, atomically replaces the two data artifacts, and replaces metadata last. The metadata records the
|
| 132 |
+
runtime and request plus SHA-256 digests of the final WAV and NPY, so an interrupted mixed-generation bundle cannot
|
| 133 |
+
validate. It records `"wav_reencoding_performed": false` because this is an internal generation capture, not WAV
|
| 134 |
+
analysis.
|
| 135 |
+
|
| 136 |
+
For the final canonical rerun, the WAV SHA-256 is
|
| 137 |
+
`e52c8884acdfbe9a71badac1ef410a5e5b23512dfc3d56a6f76228adaae87e11`, the NPY SHA-256 is
|
| 138 |
+
`42de27f795e8a1f1b97ae85fd3a540d432a8b178fdc63f7037bf0eea08c61350`, and the metadata JSON SHA-256 is
|
| 139 |
+
`7232d979f21cdf769e96c047b11ee31bbe28a9ee30ec6cd6960455c44b71aa02`. Metadata embeds the first two digests
|
| 140 |
+
along with capture-script SHA-256 `e4b6ecb87652deaa987a722e7c281aec2196295a717066aec05ef195361c0f1f`
|
| 141 |
+
and relevant Diffusers encoder-source SHA-256
|
| 142 |
+
`247ca4492a531c6593485b9fb11503da8607d03315c2f8b313dc6ae44985336b`.
|
| 143 |
+
|
| 144 |
+
This run proves that the documented token shape, ranges, frame rate, priming alignment, and generation/rendering path
|
| 145 |
+
are executable. It does **not** supply the missing WAV-to-tokenizer path. Calling `encode_audio()` for the generated
|
| 146 |
+
WAV is still blocked by the released DAV capabilities, so there is no recovered token trajectory to compare.
|
| 147 |
+
Per-codebook WAV re-encoding agreement percentages therefore remain unavailable; comparing
|
| 148 |
+
`music3_internal_tokens.npy` with itself would only be a self-comparison and must not be reported as re-encoding
|
| 149 |
+
agreement.
|
| 150 |
+
|
| 151 |
+
## What conversion omits
|
| 152 |
+
|
| 153 |
+
The Diffusers converter maps only `dec_in_proj.*` and `decoder.*` from `dav.pth`. Its explicit conversion dictionary
|
| 154 |
+
and decoder loop leave 427 of the 548 tensors unmapped: all 119 `encoder.*`, both `mean_proj.*`, both `logs_proj.*`,
|
| 155 |
+
and all 304 `flow.*` tensors. SGLang-Omni makes the same decoder-only selection with prefixes
|
| 156 |
+
`("dec_in_proj.", "decoder.")`. That omission is real, but restoring those keys would expose only continuous DAV
|
| 157 |
+
analysis/posterior operations. It would not restore the absent RVQ quantizer/codebooks.
|
| 158 |
+
|
| 159 |
+
## Answers to the six requested questions
|
| 160 |
+
|
| 161 |
+
1. **Is the original audio encoder present?** A waveform analysis encoder is present as `encoder.*`, but the native
|
| 162 |
+
discrete/RVQ tokenizer encoder is not. Its outputs feed continuous `mean_proj`/`logs_proj` and `flow` modules.
|
| 163 |
+
2. **Is the RVQ quantizer present?** No. There are zero matching quantizer/RVQ/VQ parameters and no acoustic
|
| 164 |
+
codebooks in `dav.pth`. The Qwen local “RVQ depth decoder” predicts residual token IDs; it is not a waveform
|
| 165 |
+
quantizer.
|
| 166 |
+
3. **Can arbitrary WAV audio be converted into valid Music 3 tokens?** No, not with the released weights. Calling
|
| 167 |
+
`encode_audio()` raises `NativeTokenizerUnavailableError` before opening the WAV and never invents token IDs.
|
| 168 |
+
4. **Do re-encoded tokens match Music 3 internal tokens?** This experiment is blocked because no re-encoded tokens
|
| 169 |
+
can be produced. Reporting per-codebook percentages would be fabricated. `native_token_compatibility.py` performs
|
| 170 |
+
strict range/layout validation and exact per-codebook comparison once both real trajectories exist.
|
| 171 |
+
5. **Can those tokens seed continuation?** No token trajectory can be recovered, so continuation cannot be seeded.
|
| 172 |
+
The current released inference entry points also start from text plus `<|audio_start|>` rather than accepting an
|
| 173 |
+
existing token/LM-cache history. Even if a tokenizer is released later, continuation must replay each complete
|
| 174 |
+
eight-code frame through the exact feedback embedding/depth-hidden path and preserve the priming alignment.
|
| 175 |
+
6. **What is missing?** The compatible waveform-to-code encoder/quantizer implementation, its RVQ acoustic codebook
|
| 176 |
+
weights and config, plus an official existing-audio history/prefill interface. Nothing here justifies retraining or
|
| 177 |
+
substituting a generic codec.
|
| 178 |
+
|
| 179 |
+
There is intentionally no `continue_audio.py`: without native tokens, such a CLI could only ignore the input audio,
|
| 180 |
+
mislabel continuous latents as tokens, or use an incompatible replacement codec. All three would violate the stated
|
| 181 |
+
conditioning contract. Likewise, a token round trip is impossible: Music 3 renders `frame_hiddens` through a
|
| 182 |
+
flow-matching model and DAV, not discrete codes directly through DAV.
|
| 183 |
+
|
| 184 |
+
## Reproducible checks
|
| 185 |
+
|
| 186 |
+
The inspection, comparison, blocked-encode, and test paths are CPU-only. Checkpoint loading uses
|
| 187 |
+
`torch.load(..., weights_only=True, map_location="cpu")`. The optional `capture_generated_tokens.py` command is the
|
| 188 |
+
explicit exception: it loads the pinned official generation runtime and uses the requested device (CUDA by default)
|
| 189 |
+
only to observe internally generated tokens. `pyproject.toml` requires `torch>=2.10.0`, after the vulnerable
|
| 190 |
+
versions listed in [GHSA-63cw-57p8-fm3p](https://github.com/advisories/GHSA-63cw-57p8-fm3p).
|
| 191 |
+
|
| 192 |
+
Install the optional generation-capture runtime with `pip install -e '.[capture]'`; that extra pins the audited
|
| 193 |
+
Diffusers Git revision and explicitly declares Accelerate, Hugging Face Hub, Transformers, and SoundFile in addition
|
| 194 |
+
to the base NumPy and Torch requirements.
|
| 195 |
+
|
| 196 |
+
```bash
|
| 197 |
+
python inspect_dav.py /path/to/dav.pth --sha256
|
| 198 |
+
python inspect_dav.py /path/to/dav.pth --json
|
| 199 |
+
|
| 200 |
+
# Both commands return nonzero and say BLOCKED before opening missing.wav.
|
| 201 |
+
python encode_audio.py missing.wav --dav /path/to/dav.pth --json
|
| 202 |
+
python round_trip_test.py missing.wav reconstructed.wav --dav /path/to/dav.pth --json
|
| 203 |
+
|
| 204 |
+
# Exact comparison, only when genuine internal and re-encoded trajectories exist.
|
| 205 |
+
python native_token_compatibility.py internal.npy recovered.npy \
|
| 206 |
+
--reference-layout frames_first --recovered-layout frames_first --json
|
| 207 |
+
|
| 208 |
+
# Capture tokens from generation itself; this does not re-encode the saved WAV.
|
| 209 |
+
python capture_generated_tokens.py \
|
| 210 |
+
--prompt "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass." \
|
| 211 |
+
--lyrics "[instrumental]" --audio-duration 1 --num-inference-steps 30 --seed 7 \
|
| 212 |
+
--local-files-only \
|
| 213 |
+
--output-wav /path/to/music3_generated_1s.wav \
|
| 214 |
+
--output-tokens /path/to/music3_internal_tokens.npy \
|
| 215 |
+
--output-metadata /path/to/music3_generation_capture.json
|
| 216 |
+
|
| 217 |
+
# Generated-sample integration is enabled only when all four are set.
|
| 218 |
+
MINIMAX_DAV_PATH=/path/to/dav.pth \
|
| 219 |
+
MINIMAX_GENERATED_WAV_PATH=/path/to/music3_generated_1s.wav \
|
| 220 |
+
MINIMAX_INTERNAL_TOKENS_PATH=/path/to/music3_internal_tokens.npy \
|
| 221 |
+
MINIMAX_GENERATION_CAPTURE_PATH=/path/to/music3_generation_capture.json \
|
| 222 |
+
pytest
|
| 223 |
+
```
|
| 224 |
+
|
| 225 |
+
Synthetic tests cover flat and nested/`62000_generator` checkpoints, capability verdicts, pre-WAV failure, explicit
|
| 226 |
+
token layouts, all codebook ranges, exact/mismatched agreement, unequal frame counts, capture ordering, metadata-last
|
| 227 |
+
replacement, source cleanliness, and partial artifact configuration. The environment-gated integration test checks
|
| 228 |
+
the real 548-tensor release, validates the full generated-capture schema, recomputes the WAV/NPY/source/script hashes,
|
| 229 |
+
and confirms the blocked encoder path. Impossible experiments--WAV token encoding, token round trip, re-encoding
|
| 230 |
+
agreement, and continuation--are explicitly reported as blocked rather than marked passed.
|
| 231 |
+
|
| 232 |
+
## Audited sources and revisions
|
| 233 |
+
|
| 234 |
+
- MiniMax release checkpoint: [MiniMaxAI/MiniMax-Music3 at revision
|
| 235 |
+
`fbdf52fbaaca799592917417eb05f1899f1255ec`](https://huggingface.co/MiniMaxAI/MiniMax-Music3/tree/fbdf52fbaaca799592917417eb05f1899f1255ec).
|
| 236 |
+
- Hugging Face Diffusers revision `90b4e34e79a86ec5e7f2437634fe95ecd2108796`: the
|
| 237 |
+
[Music 3 converter's DAV selection](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/scripts/convert_minimax_music3_to_diffusers.py#L96-L127),
|
| 238 |
+
[global/local token generation and feedback embedding](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/src/diffusers/modular_pipelines/minimax_music3/encoders.py#L102-L148),
|
| 239 |
+
[25 Hz autoregressive loop](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/src/diffusers/modular_pipelines/minimax_music3/encoders.py#L287-L358), and
|
| 240 |
+
[seven-head local depth decoder](https://github.com/huggingface/diffusers/blob/90b4e34e79a86ec5e7f2437634fe95ecd2108796/src/diffusers/models/transformers/minimax_music3_rvq_depth_decoder.py#L91-L139).
|
| 241 |
+
- SGLang-Omni revision `d0edde030334a1e54a6644ba3ab21eabc4a01f73`: exact
|
| 242 |
+
[special IDs and audio-code offset](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/prompt.py#L9-L20),
|
| 243 |
+
[eight-code feedback embedding](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/sglang_model.py#L82-L96),
|
| 244 |
+
[per-frame code/hidden alignment](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/model_runner.py#L252-L289), and
|
| 245 |
+
[decoder-only DAV selection](https://github.com/sgl-project/sglang-omni/blob/d0edde030334a1e54a6644ba3ab21eabc4a01f73/sglang_omni/models/minimax_music3/dav.py#L114-L154).
|
LICENSE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or Derivative
|
| 95 |
+
Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work, excluding
|
| 103 |
+
those notices that do not pertain to any part of the Derivative
|
| 104 |
+
Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one of
|
| 111 |
+
the following places: within a NOTICE text file distributed as
|
| 112 |
+
part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and do
|
| 117 |
+
not modify the License. You may add Your own attribution notices
|
| 118 |
+
within Derivative Works that You distribute, alongside or as an
|
| 119 |
+
addendum to the NOTICE text from the Work, provided that such
|
| 120 |
+
additional attribution notices cannot be construed as modifying
|
| 121 |
+
the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright 2026 Music3Lab contributors
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
MODEL_CARD.md
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Model card — Music3Lab adapters & encoders
|
| 2 |
+
|
| 3 |
+
This card covers the trained checkpoints produced by Music3Lab. **The base
|
| 4 |
+
MiniMax-Music3 model is not covered here** — see its own card at
|
| 5 |
+
`MiniMaxAI/MiniMax-Music3`.
|
| 6 |
+
|
| 7 |
+
> ⚠️ **License:** these checkpoints are likely **derivative works of
|
| 8 |
+
> MiniMax-Music3** (trained through its frozen decoder and/or on its captured
|
| 9 |
+
> state). Their redistribution may be governed by the MiniMax-Music3 license.
|
| 10 |
+
> Confirm before publishing. See THIRD_PARTY.md. The checkpoints are **not** in
|
| 11 |
+
> this Git repo; only pointers are (`checkpoints/README.md`).
|
| 12 |
+
|
| 13 |
+
## Common facts
|
| 14 |
+
|
| 15 |
+
- **Base model:** MiniMax-Music3 (frozen; never fine-tuned by this project).
|
| 16 |
+
- **What stays frozen:** the DAV/Flow decoder, the Global/Local language models,
|
| 17 |
+
and the vocoder. Only the small adapters/encoders below are trained.
|
| 18 |
+
- **Training compute:** single NVIDIA H100 80 GB.
|
| 19 |
+
- **Evaluation:** preregistered objective gates only (SI-SDR, SNR, correlation,
|
| 20 |
+
reconstruction ruler, latent NMSE, loudness/stereo, boundary continuity,
|
| 21 |
+
anti-copy). **No human listening tests. No learned aesthetic reward.**
|
| 22 |
+
- **Author's own songs** were held out of all training and checkpoint selection
|
| 23 |
+
(out-of-distribution evaluation only).
|
| 24 |
+
|
| 25 |
+
## Checkpoints
|
| 26 |
+
|
| 27 |
+
| Checkpoint | Arch / trainable params | Trained on | Gate result |
|
| 28 |
+
|---|---|---|---|
|
| 29 |
+
| `flow-encoder` (base) | WAV `[B,2,44032]` → latent `[B,128,86]`, Conv1d frontend (~1M) | Music3-generated WAV/latent teacher pairs | ✅ pilot pass; ~1.34 ms one-pass |
|
| 30 |
+
| `external-finetune` encoder | same arch, fine-tuned | 542 LAION real-music + Music3 teachers | 🟡 **rejected specialist** — external held-out ruler +74.9%, SI-SDR 2.03→8.24 dB, but protected teacher latent NMSE regressed +13.2% (> 5% limit) |
|
| 31 |
+
| `masked-flow-inpaint` adapter | rank-4 LoRA on 108 QKV projs × 36 Flow blocks + 3 embeddings = 1,775,616 | captured Music3 conditions | ✅ pilot: +30.8% latent NMSE, +20.1% hole ruler (captured-condition only) |
|
| 32 |
+
| `learned-continuation` / `-residual` adapter | rank-4 Flow LoRA (~1.7M) | adjacent real-audio windows | ❌ lost to repeat/roll baselines |
|
| 33 |
+
| `native-token` adapter | 1×16384 semantic head + 7×1024 residual heads | 96 Music3 token captures | ❌ residual CE regressed; not a tokenizer |
|
| 34 |
+
| `acoustic-fim-v2` | 2-sided waveform U-Net, 4,063,090 | 542 LAION real-music | ❌ +5.9% vs +10% gate |
|
| 35 |
+
| `audio-prepend` / `waveform-right-context-prepend` | spectral/attention (~5.4M) | real-audio right-context | ❌ seam / anti-copy gates failed |
|
| 36 |
+
| `waveform-causal-continuation` | causal waveform net | real-audio history | ❌ collapsed to near-exact repeat-tail copy |
|
| 37 |
+
| `native-state-stage0` (v3) | 3-branch posterior (log-Mel + stereo STFT + latent), 8 Conformer blocks | one Music3 clip | ✅ one-clip 25/25 alignment (feedback NMSE 5.1e-13) — **interpolation, not a general encoder** |
|
| 38 |
+
| `native-state-distill` | c0 soft-logit distillation | 104 captured clips | 🟡 bounded pass (held-out hard CE 8.97); c1–c7 absent |
|
| 39 |
+
| `native-stage1-residual` | autoregressive c1–c7 heads | 1,024 captured clips | ❌ near-modal; 0 exact rows |
|
| 40 |
+
| `official_local_audio_heads` | frozen official Local heads, extracted | (extracted from base) | support artifact so native-state code runs without the 57 GB model |
|
| 41 |
+
|
| 42 |
+
Exact per-run commit hashes, selected steps, validation losses, and artifact
|
| 43 |
+
SHA-256s are in [`reports/RELEASE_STATUS.md`](reports/RELEASE_STATUS.md) and
|
| 44 |
+
[`reports/FINAL_RESULTS.md`](reports/FINAL_RESULTS.md).
|
| 45 |
+
|
| 46 |
+
## Intended use
|
| 47 |
+
|
| 48 |
+
Research and reproduction: studying continuous-latent audio representations,
|
| 49 |
+
editing in Music3's Flow-latent space, capture/replay of generation state, and
|
| 50 |
+
objective evaluation methodology — including studying the **negative** results.
|
| 51 |
+
|
| 52 |
+
## Out-of-scope / limitations
|
| 53 |
+
|
| 54 |
+
- Not a native WAV→token encoder for Music3 (that is unsolved with the released
|
| 55 |
+
weights).
|
| 56 |
+
- Continuous inversion is a slow research teacher (~208 s per 1 s), not real-time.
|
| 57 |
+
- Arbitrary-WAV continuation / inpainting / prepend **do not work** at the target
|
| 58 |
+
quality; those checkpoints are provided as reproducible negative results.
|
| 59 |
+
- No safety/aesthetic/musicality guarantees. Objective metrics ≠ musical quality.
|
NOTICE
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Music3Lab
|
| 2 |
+
Copyright 2026 Music3Lab contributors
|
| 3 |
+
|
| 4 |
+
This product includes software developed by the Music3Lab contributors.
|
| 5 |
+
|
| 6 |
+
Licensed under the Apache License, Version 2.0 (see LICENSE).
|
| 7 |
+
|
| 8 |
+
--------------------------------------------------------------------------------
|
| 9 |
+
This repository contains ONLY original source code, configs, scripts, tests,
|
| 10 |
+
documentation, and dataset *metadata* authored by the Music3Lab contributors.
|
| 11 |
+
|
| 12 |
+
It does NOT bundle, redistribute, or embed any of the following third-party
|
| 13 |
+
assets. These must be obtained by the user from their original sources under
|
| 14 |
+
their own licenses. See THIRD_PARTY.md for details.
|
| 15 |
+
|
| 16 |
+
- MiniMax-Music3 model weights (MiniMaxAI/MiniMax-Music3)
|
| 17 |
+
- The Hugging Face Diffusers library / its MiniMax-Music3 pipeline
|
| 18 |
+
- LAION-DISCO-12M audio (only Apache-2.0 metadata/IDs are referenced here;
|
| 19 |
+
no audio is included)
|
| 20 |
+
- Any YouTube-sourced audio
|
| 21 |
+
- Any third-party recordings
|
| 22 |
+
|
| 23 |
+
Trained adapter/encoder checkpoints produced by this project are distributed
|
| 24 |
+
separately (see checkpoints/README.md and MODEL_CARD.md) and may be subject to
|
| 25 |
+
the MiniMax-Music3 license as derivative artifacts. Confirm that license before
|
| 26 |
+
redistributing any checkpoint.
|
README.md
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Music3Lab
|
| 2 |
+
|
| 3 |
+
**Open research toolkit extending [MiniMax-Music3](https://huggingface.co/MiniMaxAI/MiniMax-Music3)
|
| 4 |
+
with arbitrary-audio encoding, continuation, inpainting, prepend generation,
|
| 5 |
+
prompt-free generation, and automated objective evaluation.**
|
| 6 |
+
|
| 7 |
+
Music3Lab is a reproducible, evidence-first toolkit built *around* the released
|
| 8 |
+
MiniMax-Music3 weights. It does not modify or redistribute those weights. Every
|
| 9 |
+
capability below was gated on preregistered objective metrics, and **the
|
| 10 |
+
failures are published alongside the successes** — they are the more useful part
|
| 11 |
+
of the research.
|
| 12 |
+
|
| 13 |
+
> **Honesty note.** This is a research toolkit, not a finished product. Several
|
| 14 |
+
> headline goals (native WAV→token encoding, true arbitrary-audio history-aware
|
| 15 |
+
> continuation/inpainting/prepend, direct long-form reference conditioning) were
|
| 16 |
+
> attempted and **did not pass their gates**. Those negative results, their code,
|
| 17 |
+
> configs, and exact metrics are all here on purpose.
|
| 18 |
+
|
| 19 |
+
---
|
| 20 |
+
|
| 21 |
+
## What it does
|
| 22 |
+
|
| 23 |
+
| Capability | Status | Notes / measured result |
|
| 24 |
+
|---|---|---|
|
| 25 |
+
| **Checkpoint audit** of the released weights | ✅ available | Every released tensor classified; proves there is **no** native RVQ waveform tokenizer in `dav.pth`. See [FINDINGS.md](FINDINGS.md). |
|
| 26 |
+
| **Continuous WAV → Flow-latent encoder** | ✅ available | One-pass ≈1.34 ms (base pilot). External real-music fine-tune improved held-out audio ruler 74.9% and SI-SDR 2.03→8.24 dB, but is a **rejected specialist** (protected teacher latent regressed +13.2%), not a promoted champion. |
|
| 27 |
+
| **Latent inversion** (research/oracle mode) | ✅ available | Iterative; a 1 s external clip reached 22.2 dB SI-SDR / 0.997 correlation. Slow (~208 s per 1 s) — a teacher, not a real-time encoder. |
|
| 28 |
+
| **Masked-Flow inpainting** (captured Music3 conditions) | ✅ available | +30.8% latent NMSE, +20.1% hole audio-ruler vs zero-adapter. Captured-condition only. |
|
| 29 |
+
| **Captured-state style continuation** | ✅ available | 12 s → 16 s, four candidates, objective style/seam ranking. Captured-state only, deterministic-from-frame-0. |
|
| 30 |
+
| **Full-state resume** | ✅ available | Serializes KV cache + CUDA RNG; reproduces frames + Flow chunks exactly across processes (deterministic backend). |
|
| 31 |
+
| **Reference-guided append** (CPU) | ✅ available | Appends a chosen reference-style candidate with **bit-exact** source preservation outside the crossfade. |
|
| 32 |
+
| **Prompt-free generation** | ✅ available | No user text; internal MIR/planning→text bridge. Best-of-N up to 90 s (max policy: 3/8 eligible full-length). |
|
| 33 |
+
| **Reference-style generation** | 🟡 partial | 8 s, ranked by direct continuous-latent + MIR similarity. Not direct model conditioning or full style transfer. |
|
| 34 |
+
| **Objective evaluation suite** | 🟡 partial | Integrity, reconstruction, SI-SDR/SNR, correlation, loudness/stereo, anti-copy. No learned musicality/aesthetic judges. |
|
| 35 |
+
| **Native WAV → Music3 RVQ tokens** | ❌ blocked | Released `dav.pth` has no quantizer/codebooks. (`62000_generator` is a PyTorch **ZIP folder name**, not a component.) |
|
| 36 |
+
| **Arbitrary-WAV continuation** | ❌ failed | Learned conditioner lost to repeat/roll baselines. |
|
| 37 |
+
| **Two-sided acoustic FIM (arbitrary WAV)** | ❌ failed | +5.9% ruler vs required +10%; boundaries worse than interpolation. |
|
| 38 |
+
| **Arbitrary-WAV / waveform prepend** | ❌ failed | Failed seam / anti-copy gates. |
|
| 39 |
+
| **Native-state Stage-1 residual prediction** | ❌ failed | Near-modal; mean CE 6.834, exact full token rows 0. |
|
| 40 |
+
| **Direct long-form reference conditioning** | ❌ failed | Tempo drift / early-EOS; no eligible 60 s candidate. |
|
| 41 |
+
| **Enforceable negative prompts (e.g. "no vocals")** | ❌ not enforceable | Reported honestly as `NOT_ENFORCEABLE`. |
|
| 42 |
+
|
| 43 |
+
`music3lab status --json` is the authoritative machine-readable capability
|
| 44 |
+
matrix. Full write-ups are in [`reports/`](reports/) and
|
| 45 |
+
[`reports/FINAL_RESULTS.md`](reports/FINAL_RESULTS.md).
|
| 46 |
+
|
| 47 |
+
---
|
| 48 |
+
|
| 49 |
+
## What it is NOT
|
| 50 |
+
|
| 51 |
+
- ❌ It does **not** include MiniMax-Music3 weights. You download those yourself.
|
| 52 |
+
- ❌ It does **not** redistribute any audio — not LAION/YouTube audio, not the
|
| 53 |
+
author's own songs. Only dataset *metadata* (IDs, hashes, splits) is included.
|
| 54 |
+
- ❌ It is **not** a native audio tokenizer for Music3. That does not exist in
|
| 55 |
+
the public release (see [FINDINGS.md](FINDINGS.md)).
|
| 56 |
+
|
| 57 |
+
---
|
| 58 |
+
|
| 59 |
+
## Quickstart
|
| 60 |
+
|
| 61 |
+
```bash
|
| 62 |
+
# 1. Environment (Python 3.10; exact pins in requirements.lock)
|
| 63 |
+
python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
|
| 64 |
+
pip install -e . # core (CPU inspection/eval)
|
| 65 |
+
pip install -e ".[capture]" # + diffusers/transformers for generation (GPU)
|
| 66 |
+
|
| 67 |
+
# 2. Get the base model yourself (NOT bundled). See REPRODUCING.md.
|
| 68 |
+
hf download MiniMaxAI/MiniMax-Music3 --local-dir ./models/minimax-music3
|
| 69 |
+
|
| 70 |
+
# 3. Inspect the released checkpoint (CPU, no GPU, no weights modified)
|
| 71 |
+
python inspect_dav.py ./models/minimax-music3/dav.pth --json --sha256
|
| 72 |
+
|
| 73 |
+
# 4. Machine-readable capability matrix
|
| 74 |
+
music3lab status --json
|
| 75 |
+
```
|
| 76 |
+
|
| 77 |
+
Full setup, model download, one-command demo, and benchmark commands:
|
| 78 |
+
[REPRODUCING.md](REPRODUCING.md).
|
| 79 |
+
|
| 80 |
+
---
|
| 81 |
+
|
| 82 |
+
## Repository layout
|
| 83 |
+
|
| 84 |
+
```
|
| 85 |
+
music3lab/
|
| 86 |
+
├── src/music3lab/ # the installable package (tested; layout preserved)
|
| 87 |
+
│ ├── codec/ # encoders, external fine-tune, native-state experiments
|
| 88 |
+
│ ├── editing/ # continuation / inpaint / prepend / append
|
| 89 |
+
│ ├── autonomous/ # champion/challenger promotion + rollback
|
| 90 |
+
│ ├── inversion*.py # latent inversion (research mode)
|
| 91 |
+
│ ├── eval.py # objective evaluation
|
| 92 |
+
│ └── ... # baseline capture, checkpoint audit, pipeline, release
|
| 93 |
+
├── configs/ # frozen experiment/training configs (34)
|
| 94 |
+
├── scripts/ # runnable training / experiment / benchmark scripts (38)
|
| 95 |
+
│ └── data/ # LAION downloader (laion_ingest.py, laion_freeze_interim.py)
|
| 96 |
+
├── tests/ # focused + adversarial suites (58)
|
| 97 |
+
├── reports/ # per-capability write-ups + FINAL_RESULTS.md
|
| 98 |
+
├── data/laion_disco/ # dataset METADATA only (IDs, hashes, splits) — no audio
|
| 99 |
+
├── checkpoints/ # POINTERS to trained adapters (no weights) — see README there
|
| 100 |
+
├── examples/ # how to reproduce demo outputs (no bundled audio)
|
| 101 |
+
├── requirements.lock # exact pinned environment
|
| 102 |
+
├── LICENSE NOTICE THIRD_PARTY.md MODEL_CARD.md DATA.md TRAINING.md REPRODUCING.md
|
| 103 |
+
```
|
| 104 |
+
|
| 105 |
+
**Training.** Every trainable component ships its training script + config +
|
| 106 |
+
tests. See [TRAINING.md](TRAINING.md) for the full table and how to push the
|
| 107 |
+
open problems (native tokenization, arbitrary-audio editing).
|
| 108 |
+
|
| 109 |
+
**Note on structure.** The conceptual grouping (encoder / continuation / inpaint
|
| 110 |
+
/ prepend / eval) is preserved *thematically* via the `codec/` and `editing/`
|
| 111 |
+
subpackages and this map, rather than by physically splitting `src/` — that keeps
|
| 112 |
+
the 48-test suite green and the package importable for a reproducible v0.1.0. A
|
| 113 |
+
physical refactor into top-level `encoder/continuation/...` packages is a
|
| 114 |
+
possible later, separately-tested change.
|
| 115 |
+
|
| 116 |
+
---
|
| 117 |
+
|
| 118 |
+
## The core finding
|
| 119 |
+
|
| 120 |
+
The released `dav.pth` is a **continuous** DAV analysis encoder + Gaussian
|
| 121 |
+
posterior heads + flow model + waveform decoder. It contains **no** RVQ/VQ
|
| 122 |
+
quantizer, **no** acoustic codebooks, and **no** `generator`/`62000_generator`
|
| 123 |
+
tensors. The string `62000_generator` is only the root folder name inside the
|
| 124 |
+
PyTorch ZIP archive — not a model component. Music3's eight-stream token space
|
| 125 |
+
therefore cannot be produced from an arbitrary WAV with the released weights.
|
| 126 |
+
Everything Music3Lab does works either in the continuous Flow-latent space or
|
| 127 |
+
from *captured* generation state. Details and reproduction: [FINDINGS.md](FINDINGS.md).
|
| 128 |
+
|
| 129 |
+
---
|
| 130 |
+
|
| 131 |
+
## Trained checkpoints & data
|
| 132 |
+
|
| 133 |
+
- **Adapters/encoders** are released separately (Hugging Face) — see
|
| 134 |
+
[checkpoints/README.md](checkpoints/README.md) and [MODEL_CARD.md](MODEL_CARD.md).
|
| 135 |
+
⚠️ They are derivatives of MiniMax-Music3 and may be governed by its license;
|
| 136 |
+
confirm before redistributing.
|
| 137 |
+
- **Dataset**: only LAION-DISCO-12M metadata + a downloader are shipped. No audio.
|
| 138 |
+
See [DATA.md](DATA.md).
|
| 139 |
+
|
| 140 |
+
---
|
| 141 |
+
|
| 142 |
+
## License
|
| 143 |
+
|
| 144 |
+
Original Music3Lab code: **Apache-2.0** ([LICENSE](LICENSE), [NOTICE](NOTICE)).
|
| 145 |
+
Third-party components and their separate licenses: [THIRD_PARTY.md](THIRD_PARTY.md).
|
REPORT_AUDIO_PREPEND.md
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Arbitrary-WAV suffix-conditioned prepend
|
| 2 |
+
|
| 3 |
+
This milestone trains a fresh, suffix-only continuous-latent projector and a
|
| 4 |
+
fresh rank-4 QKV Flow adapter on the frozen external 542/68/68 split. The
|
| 5 |
+
following one-second audio window is the only condition; the preceding window
|
| 6 |
+
is hidden and used only as the target. User-owned songs remain OOD evaluation
|
| 7 |
+
material and are not training inputs.
|
| 8 |
+
|
| 9 |
+
The output compositor is exactly:
|
| 10 |
+
|
| 11 |
+
`generated_prefix[:-T] + equal_power(generated_prefix[-T:], source[:T]) + source[T:]`
|
| 12 |
+
|
| 13 |
+
where `1 <= T <= 1024`. This is an acoustic local prepend experiment, not a
|
| 14 |
+
native-token, semantic, prompt-conditioned, or long-song prepend claim.
|
REPORT_REFERENCE_RANKED_POOL.md
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Reference-ranked retained long-form pool
|
| 2 |
+
|
| 3 |
+
Release status: **AVAILABLE_CPU_POST_GENERATION_ONLY** at implementation commit
|
| 4 |
+
`a7dd17852b0db6858aa32eb7a43c84c2e4c155f3`.
|
| 5 |
+
|
| 6 |
+
This milestone ranks only the three frozen, normalized exact-90-second retained
|
| 7 |
+
WAVs (`candidate-002`, `candidate-003`, and `candidate-007`) from the completed
|
| 8 |
+
prompt-free maximum-90-second pool. Raw renders, early-EOS rows, and unlisted
|
| 9 |
+
candidates are rejected before scoring.
|
| 10 |
+
|
| 11 |
+
The source WAV is not passed to a generator. A frozen rejected-specialist Flow
|
| 12 |
+
encoder recomputes deterministic eight-crop continuous-latent profiles on CPU,
|
| 13 |
+
and the f040 compatibility weights and hard gates select among existing WAVs.
|
| 14 |
+
No audio generation or training is performed. The runtime result is therefore
|
| 15 |
+
`reference_ranked_long_form_pool_not_direct_audio_conditioning`, not native style
|
| 16 |
+
transfer, direct audio conditioning, artistic-quality evidence, or enforceable
|
| 17 |
+
vocal exclusion.
|
| 18 |
+
|
| 19 |
+
The immutable run configuration is `configs/reference-ranked-pool-v1.yaml` and
|
| 20 |
+
the executable wrapper is `scripts/run_reference_ranked_pool.py`. The fresh
|
| 21 |
+
runtime JSON and Markdown report are published beside the selected WAV rather
|
| 22 |
+
than checked into source control.
|
| 23 |
+
|
| 24 |
+
Authoritative evidence is
|
| 25 |
+
`/home/ubuntu/minimax-reference-ranked-pool-evidence/loveonme-exact90-f040ab1`.
|
| 26 |
+
Selected WAV/manifest JSON/report SHA-256 values are
|
| 27 |
+
`d5cbd5c920745e106b14aaf6659d95255f8b0f8d73a8b9f97ab47ecc00f72207`,
|
| 28 |
+
`41725b020aadcae2d58461bd8c0ba19e00ae3ded1596d82df3e184fac9650b8b`,
|
| 29 |
+
and `44f174843c9103667b709dd41cd9d64158819b8721d183fcfe9cf6990be48075`.
|
| 30 |
+
Vocal exclusion is `NOT_ENFORCEABLE`; native negative prompting is false.
|
REPORT_REFERENCE_STYLE.md
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Reference-style candidate matching
|
| 2 |
+
|
| 3 |
+
`music3lab reference-style` generates eight independent Music3 candidates from
|
| 4 |
+
an internal text description derived only from measured tempo, key, energy, and
|
| 5 |
+
stereo width. A frozen external continuous encoder then scores deterministic
|
| 6 |
+
one-second crops from the reference and each candidate. The scorer is the
|
| 7 |
+
rejected external specialist and is explicitly restricted to
|
| 8 |
+
`REJECTED_EXTERNAL_SPECIALIST_STYLE_SCORER_ONLY`; it is neither retrained nor
|
| 9 |
+
promoted and does not produce Music3 native tokens.
|
| 10 |
+
|
| 11 |
+
Automatic selection combines direct latent-profile distance with measured MIR
|
| 12 |
+
compatibility and a bounded diversity ranking bonus. Hard technical and
|
| 13 |
+
anti-copy gates run before ranking. Creativity means candidate diversity only,
|
| 14 |
+
not artistic quality. This path is not direct model audio conditioning, full
|
| 15 |
+
semantic style transfer, guaranteed vocal exclusion, or artistic superiority.
|
| 16 |
+
|
| 17 |
+
## Measured H100 result
|
| 18 |
+
|
| 19 |
+
At commit `f040ab152908ba343ea5e8ffbdb86fae05e4c211`, the path generated eight
|
| 20 |
+
independent eight-second candidates from the user reference and selected index
|
| 21 |
+
0 / seed 101. The selected WAV SHA-256 is
|
| 22 |
+
`a4af7f05a6e22db7eb9696058f41b7a099f096415eb8212940d9bb9c58be795d`.
|
| 23 |
+
All eight candidates were scored using direct frozen continuous-latent
|
| 24 |
+
profiles, MIR compatibility, and a bounded creativity/diversity bonus.
|
| 25 |
+
Generation took 63.7525 seconds after one 4.8467-second pipeline load; peak
|
| 26 |
+
CUDA allocation was 24,467,387,392 bytes.
|
| 27 |
+
|
| 28 |
+
The `vocals` constraint was `NOT_ENFORCEABLE` and
|
| 29 |
+
`native_negative_prompt_used` was false. Evidence:
|
| 30 |
+
`/home/ubuntu/minimax-reference-style-evidence/user-loveonme-f040ab1-seed101.reference-style`
|
| 31 |
+
(manifest SHA-256
|
| 32 |
+
`2ada3a88bc57936810e7c8b8cf920d722225d438ca697745d9869ff4eeaa7485`).
|
REPRODUCING.md
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Reproducing Music3Lab
|
| 2 |
+
|
| 3 |
+
## Hardware / software used
|
| 4 |
+
|
| 5 |
+
- **GPU:** 1× NVIDIA H100 80 GB (generation/training). CPU-only is enough for
|
| 6 |
+
checkpoint inspection, the objective evaluators, and all `preflight`/`status`
|
| 7 |
+
commands.
|
| 8 |
+
- **Driver:** 580.105.08 · **CUDA-capable** build of PyTorch.
|
| 9 |
+
- **OS:** Ubuntu 22.04 (Linux). The CLI also runs on Windows for CPU paths.
|
| 10 |
+
- **Python:** 3.10.12.
|
| 11 |
+
|
| 12 |
+
Exact pins are in [`requirements.lock`](requirements.lock) (75 packages) and
|
| 13 |
+
[`environment.yml`](environment.yml). Notable: `torch==2.13.0`,
|
| 14 |
+
`transformers==5.15.0`, `accelerate==1.14.0`, `safetensors==0.8.0`,
|
| 15 |
+
`numpy==2.2.6`, `pydantic==2.13.4`.
|
| 16 |
+
|
| 17 |
+
> **Diffusers is pinned to an exact commit**, not a PyPI release, because the
|
| 18 |
+
> MiniMax-Music3 pipeline landed on `main`:
|
| 19 |
+
> `diffusers @ git+https://github.com/huggingface/diffusers.git@90b4e34e79a86ec5e7f2437634fe95ecd2108796`
|
| 20 |
+
> (already declared in `pyproject.toml`'s `[capture]` extra).
|
| 21 |
+
|
| 22 |
+
## 1. Install
|
| 23 |
+
|
| 24 |
+
```bash
|
| 25 |
+
python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
|
| 26 |
+
pip install -e . # core: inspection + eval (CPU OK)
|
| 27 |
+
pip install -e ".[capture]" # + diffusers/transformers for generation (GPU)
|
| 28 |
+
# For a byte-exact environment instead:
|
| 29 |
+
# pip install -r requirements.lock
|
| 30 |
+
```
|
| 31 |
+
|
| 32 |
+
## 2. Get the base model (NOT bundled)
|
| 33 |
+
|
| 34 |
+
```bash
|
| 35 |
+
hf download MiniMaxAI/MiniMax-Music3 --local-dir ./models/minimax-music3
|
| 36 |
+
```
|
| 37 |
+
|
| 38 |
+
Music3Lab never modifies these files. `models/` is git-ignored.
|
| 39 |
+
|
| 40 |
+
## 3. One-command checks (CPU, no weights modified)
|
| 41 |
+
|
| 42 |
+
```bash
|
| 43 |
+
# The core finding: no RVQ tokenizer in the released checkpoint
|
| 44 |
+
python inspect_dav.py ./models/minimax-music3/dav.pth --json --sha256
|
| 45 |
+
|
| 46 |
+
# Machine-readable capability matrix (the source of truth for what works)
|
| 47 |
+
music3lab status --json
|
| 48 |
+
|
| 49 |
+
# Discover exact subcommands and their flags
|
| 50 |
+
music3lab --help
|
| 51 |
+
```
|
| 52 |
+
|
| 53 |
+
## 4. Demo generation (GPU + [capture] extra + downloaded model)
|
| 54 |
+
|
| 55 |
+
Use `music3lab --help` for the authoritative subcommand list and flags. The
|
| 56 |
+
demonstrated, passing generation/edit paths are:
|
| 57 |
+
|
| 58 |
+
- `music3lab` prompt-free best-of-N generation (up to 90 s)
|
| 59 |
+
- `music3lab` reference-style ranked generation (8 s)
|
| 60 |
+
- `music3lab captured-style-continue` (captured-state continuation, 12 s → 16 s)
|
| 61 |
+
- `music3lab reference-guided-append` (CPU append with exact source preservation)
|
| 62 |
+
- `music3lab reference-ranked-pool` (CPU ranking of a retained pool)
|
| 63 |
+
|
| 64 |
+
Each writes a WAV plus a JSON sidecar and a `report.md` with the objective
|
| 65 |
+
metrics and artifact SHA-256s. Preflight variants (`music3lab preflight ...`) do
|
| 66 |
+
static identity checks with **no GPU and no model load**.
|
| 67 |
+
|
| 68 |
+
## 5. Benchmarks / evidence
|
| 69 |
+
|
| 70 |
+
Per-capability numbers, gates, selected steps, and artifact hashes are recorded
|
| 71 |
+
in [`reports/`](reports/) — start with
|
| 72 |
+
[`reports/RELEASE_STATUS.md`](reports/RELEASE_STATUS.md) and
|
| 73 |
+
[`reports/FINAL_RESULTS.md`](reports/FINAL_RESULTS.md). These are the frozen
|
| 74 |
+
measured results, not marketing claims.
|
| 75 |
+
|
| 76 |
+
## 6. Regenerating training data / corpora
|
| 77 |
+
|
| 78 |
+
- Music3 teacher pairs (WAV/latent/token captures): `scripts/` capture runners,
|
| 79 |
+
which load the model you downloaded in step 2.
|
| 80 |
+
- External LAION corpus: `scripts/data/laion_ingest.py` then
|
| 81 |
+
`scripts/data/laion_freeze_interim.py` (adjust the hardcoded `ROOT` paths
|
| 82 |
+
first; see [DATA.md](DATA.md) and its ToS caveat).
|
| 83 |
+
|
| 84 |
+
## Notes on determinism
|
| 85 |
+
|
| 86 |
+
Exact cross-process reproduction (full-state resume, captured-state continuation)
|
| 87 |
+
requires `torch.use_deterministic_algorithms(True)` **from frame 0**. The older
|
| 88 |
+
Phase-0 captures were produced under the nondeterministic SDPA backend and cannot
|
| 89 |
+
be bit-exactly resumed — a documented limitation, not a bug. See
|
| 90 |
+
[`docs/CAPTURED_STATE_STYLE_CONTINUATION.md`](docs/CAPTURED_STATE_STYLE_CONTINUATION.md).
|
capture_generated_tokens.py
ADDED
|
@@ -0,0 +1,460 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Capture native token rows produced internally by official Music 3 generation.
|
| 2 |
+
|
| 3 |
+
This utility observes the existing autoregressive path. It does not encode or
|
| 4 |
+
re-encode a WAV and cannot turn arbitrary audio into Music 3 tokens.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
import argparse
|
| 10 |
+
import functools
|
| 11 |
+
import hashlib
|
| 12 |
+
import importlib
|
| 13 |
+
import json
|
| 14 |
+
import os
|
| 15 |
+
import subprocess
|
| 16 |
+
import tempfile
|
| 17 |
+
from contextlib import contextmanager
|
| 18 |
+
from dataclasses import dataclass, field
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
from typing import Any, Callable, Iterator
|
| 21 |
+
|
| 22 |
+
import numpy as np
|
| 23 |
+
|
| 24 |
+
FRAME_RATE_HZ = 25
|
| 25 |
+
NUM_CODEBOOKS = 8
|
| 26 |
+
CODEBOOK_SIZES = (16_384, 1_024, 1_024, 1_024, 1_024, 1_024, 1_024, 1_024)
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def _validate_token_rows(value: Any) -> np.ndarray:
|
| 30 |
+
array = np.asarray(value)
|
| 31 |
+
if array.ndim != 2 or array.shape[1] != NUM_CODEBOOKS:
|
| 32 |
+
raise ValueError(f"tokens must have shape [frames, 8], got {array.shape}")
|
| 33 |
+
if (
|
| 34 |
+
not np.issubdtype(array.dtype, np.integer)
|
| 35 |
+
or np.issubdtype(array.dtype, np.bool_)
|
| 36 |
+
):
|
| 37 |
+
raise TypeError(f"tokens must use an integer dtype, got {array.dtype}")
|
| 38 |
+
if array.shape[0] == 0:
|
| 39 |
+
raise ValueError("token trajectory must contain at least one frame")
|
| 40 |
+
normalized = np.ascontiguousarray(array, dtype=np.int64)
|
| 41 |
+
for index, size in enumerate(CODEBOOK_SIZES):
|
| 42 |
+
values = normalized[:, index]
|
| 43 |
+
if int(values.min()) < 0 or int(values.max()) >= size:
|
| 44 |
+
raise ValueError(
|
| 45 |
+
f"codebook c{index} values must be in [0, {size - 1}]"
|
| 46 |
+
)
|
| 47 |
+
return normalized
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
DEFAULT_MODEL = "MiniMaxAI/MiniMax-Music3"
|
| 51 |
+
DEFAULT_MODEL_REVISION = "fbdf52fbaaca799592917417eb05f1899f1255ec"
|
| 52 |
+
DEFAULT_DIFFUSERS_REVISION = "90b4e34e79a86ec5e7f2437634fe95ecd2108796"
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def _as_cpu_numpy(value: Any) -> np.ndarray:
|
| 56 |
+
if hasattr(value, "detach"):
|
| 57 |
+
value = value.detach()
|
| 58 |
+
if hasattr(value, "cpu"):
|
| 59 |
+
value = value.cpu()
|
| 60 |
+
if hasattr(value, "numpy"):
|
| 61 |
+
value = value.numpy()
|
| 62 |
+
return np.asarray(value)
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
@dataclass
|
| 66 |
+
class DepthCodeCapture:
|
| 67 |
+
"""Record official depth-decoder outputs while preserving call behavior."""
|
| 68 |
+
|
| 69 |
+
priming_calls: int = 1
|
| 70 |
+
_rows: list[np.ndarray] = field(default_factory=list)
|
| 71 |
+
|
| 72 |
+
@property
|
| 73 |
+
def captured_calls(self) -> int:
|
| 74 |
+
return len(self._rows)
|
| 75 |
+
|
| 76 |
+
def wrap(self, original: Callable[..., Any]) -> Callable[..., Any]:
|
| 77 |
+
@functools.wraps(original)
|
| 78 |
+
def captured(*args: Any, **kwargs: Any) -> Any:
|
| 79 |
+
result = original(*args, **kwargs)
|
| 80 |
+
if not isinstance(result, (tuple, list)) or not result:
|
| 81 |
+
raise RuntimeError("official _generate_depth_codes returned no frame-code tensor")
|
| 82 |
+
paired = _as_cpu_numpy(result[0])
|
| 83 |
+
if paired.shape != (2, NUM_CODEBOOKS):
|
| 84 |
+
raise RuntimeError(
|
| 85 |
+
"official _generate_depth_codes must return paired [2, 8] codes, "
|
| 86 |
+
f"got {paired.shape}"
|
| 87 |
+
)
|
| 88 |
+
if not np.array_equal(paired[0], paired[1]):
|
| 89 |
+
raise RuntimeError("conditional/unconditional depth-code rows are not identical")
|
| 90 |
+
row = _validate_token_rows(paired[:1])[0]
|
| 91 |
+
self._rows.append(row.copy())
|
| 92 |
+
return result
|
| 93 |
+
|
| 94 |
+
return captured
|
| 95 |
+
|
| 96 |
+
def emitted_tokens(self) -> np.ndarray:
|
| 97 |
+
if self.priming_calls < 0:
|
| 98 |
+
raise ValueError("priming_calls must be non-negative")
|
| 99 |
+
if self.captured_calls <= self.priming_calls:
|
| 100 |
+
raise RuntimeError(
|
| 101 |
+
f"captured {self.captured_calls} call(s), not enough to skip "
|
| 102 |
+
f"{self.priming_calls} priming call(s)"
|
| 103 |
+
)
|
| 104 |
+
tokens = np.stack(self._rows[self.priming_calls :], axis=0)
|
| 105 |
+
return _validate_token_rows(tokens)
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
@contextmanager
|
| 109 |
+
def capture_official_depth_codes() -> Iterator[DepthCodeCapture]:
|
| 110 |
+
"""Temporarily patch the official Diffusers depth-code function."""
|
| 111 |
+
|
| 112 |
+
encoders = importlib.import_module(
|
| 113 |
+
"diffusers.modular_pipelines.minimax_music3.encoders"
|
| 114 |
+
)
|
| 115 |
+
original = encoders._generate_depth_codes
|
| 116 |
+
capture = DepthCodeCapture(priming_calls=1)
|
| 117 |
+
patched = capture.wrap(original)
|
| 118 |
+
encoders._generate_depth_codes = patched
|
| 119 |
+
try:
|
| 120 |
+
yield capture
|
| 121 |
+
finally:
|
| 122 |
+
encoders._generate_depth_codes = original
|
| 123 |
+
|
| 124 |
+
|
| 125 |
+
def build_capture_metadata(
|
| 126 |
+
*,
|
| 127 |
+
tokens: Any,
|
| 128 |
+
audio: Any,
|
| 129 |
+
captured_calls: int,
|
| 130 |
+
priming_rows_skipped: int,
|
| 131 |
+
model_id: str,
|
| 132 |
+
requested_model_revision: str,
|
| 133 |
+
resolved_model_commit: str,
|
| 134 |
+
diffusers_identity: dict[str, Any],
|
| 135 |
+
device: str,
|
| 136 |
+
dtype: str,
|
| 137 |
+
prompt: str,
|
| 138 |
+
lyrics: str,
|
| 139 |
+
requested_audio_duration_seconds: float,
|
| 140 |
+
num_inference_steps: int,
|
| 141 |
+
seed: int,
|
| 142 |
+
sampling_rate: int,
|
| 143 |
+
wav_sha256: str,
|
| 144 |
+
tokens_npy_sha256: str,
|
| 145 |
+
capture_script_sha256: str,
|
| 146 |
+
) -> dict[str, Any]:
|
| 147 |
+
"""Validate and summarize one generated-audio/internal-token capture."""
|
| 148 |
+
|
| 149 |
+
normalized = _validate_token_rows(tokens)
|
| 150 |
+
waveform = _as_cpu_numpy(audio)
|
| 151 |
+
if waveform.ndim != 2 or waveform.shape[0] != 2:
|
| 152 |
+
raise ValueError(f"generated audio must have shape [2, samples], got {waveform.shape}")
|
| 153 |
+
if sampling_rate <= 0:
|
| 154 |
+
raise ValueError(f"sampling_rate must be positive, got {sampling_rate}")
|
| 155 |
+
if captured_calls != normalized.shape[0] + priming_rows_skipped:
|
| 156 |
+
raise ValueError(
|
| 157 |
+
"capture alignment mismatch: calls must equal emitted frames plus priming rows "
|
| 158 |
+
f"({captured_calls} != {normalized.shape[0]} + {priming_rows_skipped})"
|
| 159 |
+
)
|
| 160 |
+
if diffusers_identity.get("commit") is None:
|
| 161 |
+
raise ValueError("diffusers identity must include a commit")
|
| 162 |
+
for label, digest in (
|
| 163 |
+
("WAV", wav_sha256),
|
| 164 |
+
("NPY", tokens_npy_sha256),
|
| 165 |
+
("capture script", capture_script_sha256),
|
| 166 |
+
):
|
| 167 |
+
if len(digest) != 64 or any(char not in "0123456789abcdef" for char in digest):
|
| 168 |
+
raise ValueError(f"{label} SHA-256 must be 64 lowercase hexadecimal characters")
|
| 169 |
+
return {
|
| 170 |
+
"capture_kind": "official_internal_generation_tokens",
|
| 171 |
+
"wav_reencoding_performed": False,
|
| 172 |
+
"model_id": model_id,
|
| 173 |
+
"requested_model_revision": requested_model_revision,
|
| 174 |
+
"resolved_model_commit": resolved_model_commit,
|
| 175 |
+
"checkpoint_revision": resolved_model_commit,
|
| 176 |
+
"diffusers_revision": diffusers_identity["commit"],
|
| 177 |
+
"diffusers_source_identity": dict(diffusers_identity),
|
| 178 |
+
"capture_script_sha256": capture_script_sha256,
|
| 179 |
+
"device": device,
|
| 180 |
+
"dtype": dtype,
|
| 181 |
+
"prompt": prompt,
|
| 182 |
+
"lyrics": lyrics,
|
| 183 |
+
"requested_audio_duration_seconds": float(requested_audio_duration_seconds),
|
| 184 |
+
"num_inference_steps": int(num_inference_steps),
|
| 185 |
+
"seed": int(seed),
|
| 186 |
+
"captured_calls_including_priming": int(captured_calls),
|
| 187 |
+
"priming_rows_skipped": int(priming_rows_skipped),
|
| 188 |
+
"frame_rate_hz": FRAME_RATE_HZ,
|
| 189 |
+
"token_shape_frames_first": list(normalized.shape),
|
| 190 |
+
"token_mins": normalized.min(axis=0).tolist(),
|
| 191 |
+
"token_maxs": normalized.max(axis=0).tolist(),
|
| 192 |
+
"sampling_rate": int(sampling_rate),
|
| 193 |
+
"audio_shape_channels_first": list(waveform.shape),
|
| 194 |
+
"audio_duration_seconds": waveform.shape[1] / sampling_rate,
|
| 195 |
+
"artifact_sha256": {
|
| 196 |
+
"wav": wav_sha256,
|
| 197 |
+
"tokens_npy": tokens_npy_sha256,
|
| 198 |
+
},
|
| 199 |
+
}
|
| 200 |
+
|
| 201 |
+
def _sha256_file(path: Path) -> str:
|
| 202 |
+
digest = hashlib.sha256()
|
| 203 |
+
with path.open("rb") as handle:
|
| 204 |
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
| 205 |
+
digest.update(chunk)
|
| 206 |
+
return digest.hexdigest()
|
| 207 |
+
|
| 208 |
+
|
| 209 |
+
def _resolve_model_snapshot(
|
| 210 |
+
model_id: str,
|
| 211 |
+
requested_revision: str,
|
| 212 |
+
*,
|
| 213 |
+
local_files_only: bool,
|
| 214 |
+
) -> tuple[Path, str]:
|
| 215 |
+
from huggingface_hub import hf_hub_download
|
| 216 |
+
|
| 217 |
+
# Resolving one required manifest is sufficient to bind a Hub revision to
|
| 218 |
+
# its immutable snapshot commit without requiring every optional repo file
|
| 219 |
+
# to have been cached by the component-selective official loader.
|
| 220 |
+
manifest = Path(
|
| 221 |
+
hf_hub_download(
|
| 222 |
+
repo_id=model_id,
|
| 223 |
+
filename="modular_model_index.json",
|
| 224 |
+
revision=requested_revision,
|
| 225 |
+
local_files_only=local_files_only,
|
| 226 |
+
)
|
| 227 |
+
).absolute()
|
| 228 |
+
snapshot = manifest.parent
|
| 229 |
+
resolved_commit = snapshot.name
|
| 230 |
+
is_commit = len(requested_revision) == 40 and all(
|
| 231 |
+
char in "0123456789abcdefABCDEF" for char in requested_revision
|
| 232 |
+
)
|
| 233 |
+
if is_commit and resolved_commit.lower() != requested_revision.lower():
|
| 234 |
+
raise RuntimeError(
|
| 235 |
+
"model revision mismatch: "
|
| 236 |
+
f"requested {requested_revision}, resolved {resolved_commit}"
|
| 237 |
+
)
|
| 238 |
+
return snapshot, resolved_commit
|
| 239 |
+
|
| 240 |
+
def _verified_diffusers_revision(
|
| 241 |
+
diffusers_module: Any, expected: str
|
| 242 |
+
) -> dict[str, Any]:
|
| 243 |
+
source = Path(diffusers_module.__file__).resolve()
|
| 244 |
+
roots = [source.parent, *source.parents]
|
| 245 |
+
repository = next((root for root in roots if (root / ".git").exists()), None)
|
| 246 |
+
actual = None
|
| 247 |
+
if repository is not None:
|
| 248 |
+
completed = subprocess.run(
|
| 249 |
+
["git", "-C", str(repository), "rev-parse", "HEAD"],
|
| 250 |
+
check=True,
|
| 251 |
+
capture_output=True,
|
| 252 |
+
text=True,
|
| 253 |
+
)
|
| 254 |
+
actual = completed.stdout.strip()
|
| 255 |
+
status = subprocess.run(
|
| 256 |
+
[
|
| 257 |
+
"git",
|
| 258 |
+
"-C",
|
| 259 |
+
str(repository),
|
| 260 |
+
"status",
|
| 261 |
+
"--porcelain",
|
| 262 |
+
"--untracked-files=no",
|
| 263 |
+
],
|
| 264 |
+
check=True,
|
| 265 |
+
capture_output=True,
|
| 266 |
+
text=True,
|
| 267 |
+
).stdout.strip()
|
| 268 |
+
if status:
|
| 269 |
+
raise RuntimeError(
|
| 270 |
+
"Diffusers source checkout has dirty tracked files; "
|
| 271 |
+
"refusing an unverifiable capture"
|
| 272 |
+
)
|
| 273 |
+
source_kind = "git_checkout"
|
| 274 |
+
source_location = repository
|
| 275 |
+
tracked_clean: bool | None = True
|
| 276 |
+
else:
|
| 277 |
+
from importlib import metadata as importlib_metadata
|
| 278 |
+
|
| 279 |
+
direct_url = importlib_metadata.distribution("diffusers").read_text(
|
| 280 |
+
"direct_url.json"
|
| 281 |
+
)
|
| 282 |
+
if direct_url:
|
| 283 |
+
actual = json.loads(direct_url).get("vcs_info", {}).get("commit_id")
|
| 284 |
+
source_kind = "vcs_install"
|
| 285 |
+
source_location = source.parent
|
| 286 |
+
tracked_clean = None
|
| 287 |
+
if actual is None:
|
| 288 |
+
raise RuntimeError(f"cannot verify Diffusers git revision from {source}")
|
| 289 |
+
if actual != expected:
|
| 290 |
+
raise RuntimeError(f"Diffusers revision mismatch: expected {expected}, got {actual}")
|
| 291 |
+
relevant_source = (
|
| 292 |
+
source.parent / "modular_pipelines" / "minimax_music3" / "encoders.py"
|
| 293 |
+
)
|
| 294 |
+
if not relevant_source.is_file():
|
| 295 |
+
raise RuntimeError(
|
| 296 |
+
f"cannot hash relevant Diffusers Music 3 source: {relevant_source}"
|
| 297 |
+
)
|
| 298 |
+
return {
|
| 299 |
+
"commit": actual,
|
| 300 |
+
"source_kind": source_kind,
|
| 301 |
+
"source_location": str(source_location),
|
| 302 |
+
"tracked_clean": tracked_clean,
|
| 303 |
+
"relevant_source_path": str(relevant_source),
|
| 304 |
+
"relevant_source_sha256": _sha256_file(relevant_source),
|
| 305 |
+
}
|
| 306 |
+
|
| 307 |
+
|
| 308 |
+
def _temporary_sibling(target: Path) -> Path:
|
| 309 |
+
target.parent.mkdir(parents=True, exist_ok=True)
|
| 310 |
+
descriptor, name = tempfile.mkstemp(
|
| 311 |
+
prefix=f".{target.name}.",
|
| 312 |
+
suffix=".tmp",
|
| 313 |
+
dir=target.parent,
|
| 314 |
+
)
|
| 315 |
+
os.close(descriptor)
|
| 316 |
+
return Path(name)
|
| 317 |
+
|
| 318 |
+
|
| 319 |
+
def _fsync_file(path: Path) -> None:
|
| 320 |
+
with path.open("rb") as handle:
|
| 321 |
+
os.fsync(handle.fileno())
|
| 322 |
+
|
| 323 |
+
def run_capture(args: argparse.Namespace) -> dict[str, Any]:
|
| 324 |
+
"""Load the pinned official runtime, generate audio, and save its internal codes."""
|
| 325 |
+
|
| 326 |
+
# Heavy runtime dependencies remain lazy so importing the pure capture
|
| 327 |
+
# invariants does not initialize a model or accelerator.
|
| 328 |
+
import soundfile as sf
|
| 329 |
+
import torch
|
| 330 |
+
import diffusers
|
| 331 |
+
from diffusers import ModularPipeline
|
| 332 |
+
|
| 333 |
+
diffusers_identity = _verified_diffusers_revision(
|
| 334 |
+
diffusers, args.diffusers_revision
|
| 335 |
+
)
|
| 336 |
+
snapshot_path, resolved_model_commit = _resolve_model_snapshot(
|
| 337 |
+
args.model,
|
| 338 |
+
args.model_revision,
|
| 339 |
+
local_files_only=args.local_files_only,
|
| 340 |
+
)
|
| 341 |
+
dtype = getattr(torch, args.dtype)
|
| 342 |
+
pipe = ModularPipeline.from_pretrained(str(snapshot_path))
|
| 343 |
+
pipe.load_components(dtype=dtype)
|
| 344 |
+
actual_frame_rate = float(pipe.frame_rate)
|
| 345 |
+
if actual_frame_rate != FRAME_RATE_HZ:
|
| 346 |
+
raise RuntimeError(
|
| 347 |
+
f"Music 3 frame-rate mismatch: expected {FRAME_RATE_HZ}, got {actual_frame_rate}"
|
| 348 |
+
)
|
| 349 |
+
pipe.to(args.device)
|
| 350 |
+
generator = torch.Generator(args.device).manual_seed(args.seed)
|
| 351 |
+
|
| 352 |
+
with capture_official_depth_codes() as capture:
|
| 353 |
+
audios = pipe(
|
| 354 |
+
prompt=args.prompt,
|
| 355 |
+
lyrics=args.lyrics,
|
| 356 |
+
audio_duration=args.audio_duration,
|
| 357 |
+
num_inference_steps=args.num_inference_steps,
|
| 358 |
+
generator=generator,
|
| 359 |
+
output="audios",
|
| 360 |
+
)
|
| 361 |
+
|
| 362 |
+
tokens = capture.emitted_tokens()
|
| 363 |
+
audio = _as_cpu_numpy(audios[0]).astype(np.float32, copy=False)
|
| 364 |
+
output_tokens = Path(args.output_tokens).expanduser().resolve()
|
| 365 |
+
output_wav = Path(args.output_wav).expanduser().resolve()
|
| 366 |
+
output_metadata = Path(args.output_metadata).expanduser().resolve()
|
| 367 |
+
temp_tokens = _temporary_sibling(output_tokens)
|
| 368 |
+
temp_wav = _temporary_sibling(output_wav)
|
| 369 |
+
temp_metadata = _temporary_sibling(output_metadata)
|
| 370 |
+
try:
|
| 371 |
+
with temp_tokens.open("wb") as handle:
|
| 372 |
+
np.save(handle, tokens)
|
| 373 |
+
handle.flush()
|
| 374 |
+
os.fsync(handle.fileno())
|
| 375 |
+
sf.write(
|
| 376 |
+
temp_wav,
|
| 377 |
+
audio.T,
|
| 378 |
+
int(pipe.sampling_rate),
|
| 379 |
+
format="WAV",
|
| 380 |
+
)
|
| 381 |
+
_fsync_file(temp_wav)
|
| 382 |
+
wav_sha256 = _sha256_file(temp_wav)
|
| 383 |
+
tokens_npy_sha256 = _sha256_file(temp_tokens)
|
| 384 |
+
metadata = build_capture_metadata(
|
| 385 |
+
tokens=tokens,
|
| 386 |
+
audio=audio,
|
| 387 |
+
captured_calls=capture.captured_calls,
|
| 388 |
+
priming_rows_skipped=capture.priming_calls,
|
| 389 |
+
model_id=args.model,
|
| 390 |
+
requested_model_revision=args.model_revision,
|
| 391 |
+
resolved_model_commit=resolved_model_commit,
|
| 392 |
+
diffusers_identity=diffusers_identity,
|
| 393 |
+
device=args.device,
|
| 394 |
+
dtype=args.dtype,
|
| 395 |
+
prompt=args.prompt,
|
| 396 |
+
lyrics=args.lyrics,
|
| 397 |
+
requested_audio_duration_seconds=args.audio_duration,
|
| 398 |
+
num_inference_steps=args.num_inference_steps,
|
| 399 |
+
seed=args.seed,
|
| 400 |
+
sampling_rate=int(pipe.sampling_rate),
|
| 401 |
+
wav_sha256=wav_sha256,
|
| 402 |
+
tokens_npy_sha256=tokens_npy_sha256,
|
| 403 |
+
capture_script_sha256=_sha256_file(Path(__file__).resolve()),
|
| 404 |
+
)
|
| 405 |
+
with temp_metadata.open("w", encoding="utf-8") as handle:
|
| 406 |
+
json.dump(metadata, handle, indent=2)
|
| 407 |
+
handle.write("\n")
|
| 408 |
+
handle.flush()
|
| 409 |
+
os.fsync(handle.fileno())
|
| 410 |
+
|
| 411 |
+
# Metadata is the commit record for the pair and is replaced last. If
|
| 412 |
+
# an earlier replacement is interrupted, its hashes cannot validate.
|
| 413 |
+
os.replace(temp_tokens, output_tokens)
|
| 414 |
+
os.replace(temp_wav, output_wav)
|
| 415 |
+
os.replace(temp_metadata, output_metadata)
|
| 416 |
+
return metadata
|
| 417 |
+
finally:
|
| 418 |
+
for temporary in (temp_tokens, temp_wav, temp_metadata):
|
| 419 |
+
temporary.unlink(missing_ok=True)
|
| 420 |
+
|
| 421 |
+
def build_parser() -> argparse.ArgumentParser:
|
| 422 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 423 |
+
parser.add_argument("--prompt", required=True)
|
| 424 |
+
lyrics = parser.add_mutually_exclusive_group(required=True)
|
| 425 |
+
lyrics.add_argument("--lyrics")
|
| 426 |
+
lyrics.add_argument("--lyrics-file")
|
| 427 |
+
parser.add_argument("--audio-duration", type=float, default=1.0)
|
| 428 |
+
parser.add_argument("--num-inference-steps", type=int, default=30)
|
| 429 |
+
parser.add_argument("--seed", type=int, default=7)
|
| 430 |
+
parser.add_argument("--model", default=DEFAULT_MODEL)
|
| 431 |
+
parser.add_argument("--model-revision", default=DEFAULT_MODEL_REVISION)
|
| 432 |
+
parser.add_argument("--diffusers-revision", default=DEFAULT_DIFFUSERS_REVISION)
|
| 433 |
+
parser.add_argument(
|
| 434 |
+
"--local-files-only",
|
| 435 |
+
action="store_true",
|
| 436 |
+
help="resolve the pinned model revision from the local Hub cache only",
|
| 437 |
+
)
|
| 438 |
+
parser.add_argument("--device", default="cuda")
|
| 439 |
+
parser.add_argument(
|
| 440 |
+
"--dtype",
|
| 441 |
+
choices=("bfloat16", "float16", "float32"),
|
| 442 |
+
default="bfloat16",
|
| 443 |
+
)
|
| 444 |
+
parser.add_argument("--output-wav", default="music3_generated.wav")
|
| 445 |
+
parser.add_argument("--output-tokens", default="music3_internal_tokens.npy")
|
| 446 |
+
parser.add_argument("--output-metadata", default="music3_generation_capture.json")
|
| 447 |
+
return parser
|
| 448 |
+
|
| 449 |
+
|
| 450 |
+
def main(argv: list[str] | None = None) -> int:
|
| 451 |
+
args = build_parser().parse_args(argv)
|
| 452 |
+
if args.lyrics_file:
|
| 453 |
+
args.lyrics = Path(args.lyrics_file).expanduser().read_text(encoding="utf-8")
|
| 454 |
+
metadata = run_capture(args)
|
| 455 |
+
print(json.dumps(metadata, indent=2))
|
| 456 |
+
return 0
|
| 457 |
+
|
| 458 |
+
|
| 459 |
+
if __name__ == "__main__":
|
| 460 |
+
raise SystemExit(main())
|
checkpoints/README.md
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Checkpoints (pointers only — no weights in Git)
|
| 2 |
+
|
| 3 |
+
This directory intentionally contains **no model weights**. Trained
|
| 4 |
+
adapters/encoders are distributed separately (Hugging Face) and are pulled in by
|
| 5 |
+
you. See [../MODEL_CARD.md](../MODEL_CARD.md) for what each one is and its gate
|
| 6 |
+
result.
|
| 7 |
+
|
| 8 |
+
> ⚠️ **Before any of these are uploaded/redistributed:** confirm the
|
| 9 |
+
> MiniMax-Music3 license permits derivative-weight redistribution (see
|
| 10 |
+
> [../THIRD_PARTY.md](../THIRD_PARTY.md)). These checkpoints were trained through
|
| 11 |
+
> the frozen Music3 decoder / on captured Music3 state and may be governed by it.
|
| 12 |
+
|
| 13 |
+
## Expected layout once downloaded
|
| 14 |
+
|
| 15 |
+
Place downloaded checkpoints here (git-ignored) or point configs at them:
|
| 16 |
+
|
| 17 |
+
```
|
| 18 |
+
checkpoints/
|
| 19 |
+
├── flow-encoder/encoder.safetensors # base continuous encoder (~1.34 ms)
|
| 20 |
+
├── external-finetune/encoder.safetensors # rejected real-music specialist
|
| 21 |
+
├── masked-flow-inpaint/adapter.safetensors # captured-condition inpaint pilot
|
| 22 |
+
├── native-state-stage0-v3/native_state_stage0.safetensors
|
| 23 |
+
├── native-state-distill/native_state_distilled.safetensors
|
| 24 |
+
└── ... # failed-experiment adapters (see MODEL_CARD)
|
| 25 |
+
```
|
| 26 |
+
|
| 27 |
+
## Downloading (once the HF repo exists and license is confirmed)
|
| 28 |
+
|
| 29 |
+
```bash
|
| 30 |
+
hf download <your-org>/music3lab-adapters --local-dir ./checkpoints
|
| 31 |
+
```
|
| 32 |
+
|
| 33 |
+
`<your-org>/music3lab-adapters` is a placeholder — replace with the real
|
| 34 |
+
Hugging Face repo id after the license review. Until then, this directory is a
|
| 35 |
+
pointer only.
|
configs/acoustic-fim-v2.yaml
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.acoustic-fim-v2-config.v1
|
| 2 |
+
capability: arbitrary_waveform_short_acoustic_fim_not_native_tokens_not_semantic_fim
|
| 3 |
+
seed: 260816
|
| 4 |
+
steps: 12000
|
| 5 |
+
validation_interval: 500
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
samples_per_frame: 512
|
| 8 |
+
canvas_samples: 98304
|
| 9 |
+
left_context_samples: 32768
|
| 10 |
+
right_context_samples: 32768
|
| 11 |
+
source_endpoint_guard_samples: 220500
|
| 12 |
+
hole_frames: [16, 32]
|
| 13 |
+
model_channels: [48, 64, 96, 128, 160, 192]
|
| 14 |
+
bottleneck_dilations: [1, 2, 4, 8, 16, 32, 64, 128, 256, 512]
|
| 15 |
+
benchmark_batch_sizes_h100: [24, 32]
|
| 16 |
+
batch_24gb: 4
|
| 17 |
+
accumulation_24gb: 6
|
| 18 |
+
max_lr: 0.0002
|
| 19 |
+
min_lr: 0.00002
|
| 20 |
+
weight_decay: 0.01
|
| 21 |
+
gradient_clip_norm: 1.0
|
| 22 |
+
validation_namespace: music3lab.acoustic-fim-v2.validation.v1
|
| 23 |
+
heldout_namespace: music3lab.acoustic-fim-v2.heldout.v1
|
| 24 |
+
manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 25 |
+
splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
| 26 |
+
loss_weights:
|
| 27 |
+
hole_l1: 1.0
|
| 28 |
+
hole_l2: 0.5
|
| 29 |
+
mrstft: 0.25
|
| 30 |
+
complex_stft_nmse: 0.15
|
| 31 |
+
mid_side: 0.10
|
| 32 |
+
boundary_waveform_first_difference: 0.20
|
| 33 |
+
nonsilent_rms_floor: 0.0001
|
| 34 |
+
noncopy_correlation_threshold: 0.995
|
| 35 |
+
resampler: scipy.signal.resample_poly Kaiser-5.0 deterministic CPU
|
| 36 |
+
augmentations:
|
| 37 |
+
gain_db: [-3.0, 3.0]
|
| 38 |
+
stereo_balance_db: [-1.5, 1.5]
|
| 39 |
+
shared_polarity_probability: 0.5
|
| 40 |
+
channel_swap_probability: 0.5
|
| 41 |
+
forbidden: [side_shift, noise, resample]
|
configs/audio-continuation-v1.yaml
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.audio-continuation-config.v1
|
| 2 |
+
capability: continuous-latent local continuation; not native AR/token continuation or long-song generalization
|
| 3 |
+
sample_rate: 44100
|
| 4 |
+
source_samples: 44032
|
| 5 |
+
append_samples: 44032
|
| 6 |
+
overlap_samples: 1024
|
| 7 |
+
seed: 101
|
| 8 |
+
condition_repeat: 16
|
| 9 |
+
condition_weight: 0.95
|
| 10 |
+
temporal_context_weight: 0.05
|
| 11 |
+
fixed_noise_weight: 0.005
|
| 12 |
+
unrelated_condition: reverse_frames_and_features
|
| 13 |
+
reference_metric: frozen_five_term_audio_ruler_lower_is_better
|
| 14 |
+
continuity_metric: join_edge_rms_log_error_lower_is_better
|
| 15 |
+
zero_condition_baseline: true
|
| 16 |
+
unrelated_condition_baseline: true
|
| 17 |
+
strict_improvement_over_both_baselines: true
|
| 18 |
+
baseline_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
|
| 19 |
+
specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 20 |
+
specialist_source_gate: REJECTED_teacher_latent_nmse_regression
|
| 21 |
+
specialist_champion_eligible: false
|
| 22 |
+
text_input_allowed: false
|
| 23 |
+
captured_condition_allowed: false
|
| 24 |
+
native_tokens_allowed: false
|
| 25 |
+
future_audio_allowed: false
|
configs/audio-fim-v1.yaml
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.audio-fim-config.v1
|
| 2 |
+
steps: 1200
|
| 3 |
+
validation_interval: 100
|
| 4 |
+
euler_steps: 30
|
| 5 |
+
guidance_scale: 1.7
|
| 6 |
+
seed: 260816
|
| 7 |
+
noise_seed: 84119
|
| 8 |
+
benchmark_batch_sizes: [4, 8, 16]
|
| 9 |
+
max_lr: 0.00005
|
| 10 |
+
min_lr: 0.000005
|
| 11 |
+
hole_frames: [16, 32]
|
| 12 |
+
minimum_context_frames_each_side: 16
|
| 13 |
+
sample_rate: 44100
|
| 14 |
+
samples_per_frame: 512
|
| 15 |
+
external_encoder_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 16 |
+
external_encoder_role: REJECTED_EXTERNAL_SPECIALIST_NOT_CHAMPION
|
| 17 |
+
corpus_manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 18 |
+
corpus_splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
configs/audio-prepend-v1.yaml
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.audio-prepend-config.v1
|
| 2 |
+
steps: 1200
|
| 3 |
+
validation_interval: 100
|
| 4 |
+
euler_steps: 30
|
| 5 |
+
guidance_scale: 1.7
|
| 6 |
+
seed: 260816
|
| 7 |
+
noise_seed: 94119
|
| 8 |
+
benchmark_batch_sizes: [4, 8, 16]
|
| 9 |
+
max_lr: 0.00005
|
| 10 |
+
min_lr: 0.000005
|
| 11 |
+
transition_samples: 1024
|
| 12 |
+
sample_rate: 44100
|
| 13 |
+
samples_per_window: 44032
|
| 14 |
+
baselines: [zero_condition, unrelated_suffix, repeat_future, roll_future, silence]
|
| 15 |
+
external_encoder_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 16 |
+
external_encoder_role: REJECTED_EXTERNAL_SPECIALIST_NOT_CHAMPION
|
| 17 |
+
corpus_manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 18 |
+
corpus_splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
configs/autonomous-interim678-v1.yaml
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.autonomous-round.interim678.v1
|
| 2 |
+
capability: continuous_flow_encoder_reconstruction_only_not_native_tokens
|
| 3 |
+
suite_label: sealed_frozen_not_secret
|
| 4 |
+
split_seed: 16082026
|
| 5 |
+
bootstrap_seed: 41
|
| 6 |
+
bootstrap_samples: 2000
|
| 7 |
+
teacher_latent_nmse_maximum_regression_fraction: 0.05
|
| 8 |
+
protected_regression_maximum_fraction: 0.0
|
| 9 |
+
technical:
|
| 10 |
+
finite_required: true
|
| 11 |
+
maximum_absolute_audio: 1.0
|
| 12 |
+
selection_split: validation
|
| 13 |
+
promotion_splits: [public, sealed]
|
| 14 |
+
user22_excluded: true
|
| 15 |
+
training: false
|
| 16 |
+
candidates:
|
| 17 |
+
champion_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
|
| 18 |
+
specialist_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 19 |
+
zero_latent: synthetic_control
|
| 20 |
+
train_lookup: synthetic_control_forbidden_from_promotion
|
configs/baseline.yaml
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.baseline-config.v3
|
| 2 |
+
model_id: MiniMaxAI/MiniMax-Music3
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
diffusers_source_root: ../diffusers-music3
|
| 7 |
+
frozen_base_root: models/base/minimax-music3
|
| 8 |
+
frozen_manifest_root: manifests/base
|
| 9 |
+
artifacts_root: artifacts/baseline
|
| 10 |
+
report_path: reports/BASELINE.md
|
| 11 |
+
device: cuda
|
| 12 |
+
dtype: bfloat16
|
| 13 |
+
shard_budget_mib: 128
|
| 14 |
+
cases:
|
| 15 |
+
- name: short_parity
|
| 16 |
+
prompt:
|
| 17 |
+
provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
|
| 18 |
+
inline: "Genre: acoustic pop. BPM: 96. Key: C major. Warm and intimate, building gently into the chorus. Vocals: soft female lead, close and breathy, light stacked harmonies in the chorus. Arrangement: fingerpicked guitar and soft piano; brushed drums and upright bass enter in the chorus."
|
| 19 |
+
lyrics:
|
| 20 |
+
provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
|
| 21 |
+
inline: |-
|
| 22 |
+
[verse]
|
| 23 |
+
Morning light filtering through the pine
|
| 24 |
+
Every quiet street is yours and mine
|
| 25 |
+
[chorus]
|
| 26 |
+
Softly the world begins to breathe
|
| 27 |
+
audio_duration_seconds: 1.0
|
| 28 |
+
num_inference_steps: 30
|
| 29 |
+
ar_seed: 7
|
| 30 |
+
rng_policy: shared_stock
|
| 31 |
+
guidance_scale: 1.7
|
| 32 |
+
instrumented: false
|
| 33 |
+
requires_stock_capture_parity: true
|
| 34 |
+
- name: multichunk_acceptance
|
| 35 |
+
prompt:
|
| 36 |
+
provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
|
| 37 |
+
inline: "Genre: acoustic pop. BPM: 96. Key: C major. Warm and intimate, building gently into the chorus. Vocals: soft female lead, close and breathy, light stacked harmonies in the chorus. Arrangement: fingerpicked guitar and soft piano; brushed drums and upright bass enter in the chorus."
|
| 38 |
+
lyrics:
|
| 39 |
+
provenance: "Diffusers official MiniMax Music 3 example, docs/source/en/api/pipelines/minimax_music3.md at 90b4e34e79a86ec5e7f2437634fe95ecd2108796"
|
| 40 |
+
inline: |-
|
| 41 |
+
[verse]
|
| 42 |
+
Morning light filtering through the pine
|
| 43 |
+
Every quiet street is yours and mine
|
| 44 |
+
[chorus]
|
| 45 |
+
Softly the world begins to breathe
|
| 46 |
+
audio_duration_seconds: 12.0
|
| 47 |
+
num_inference_steps: 30
|
| 48 |
+
ar_seed: 7
|
| 49 |
+
rng_policy: shared_stock
|
| 50 |
+
guidance_scale: 1.7
|
| 51 |
+
instrumented: true
|
| 52 |
+
requires_multichunk_acceptance: true
|
configs/captured-state-style-continuation-v1.yaml
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.captured-state-style-continuation.v1
|
| 2 |
+
capability: captured_state_same_style_continuation_only
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 5 |
+
base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 6 |
+
execution_policy: production_deterministic_from_frame0
|
| 7 |
+
dtype: bfloat16
|
| 8 |
+
root_seed: 7007
|
| 9 |
+
candidates: 4
|
| 10 |
+
prefix_frames: 300
|
| 11 |
+
new_frames: 100
|
| 12 |
+
sample_rate: 44100
|
| 13 |
+
source_samples: 529200
|
| 14 |
+
output_samples: 705600
|
| 15 |
+
guard_samples: 2048
|
| 16 |
+
flow_chunk_starts: [0, 100, 200]
|
| 17 |
+
inference_steps: 30
|
| 18 |
+
checkpoint_state_sha256: 848fff52c474b97aac46616cdaf4f67370212c95872fa4a660220689c7789f80
|
| 19 |
+
prefix_fused_bundle_sha256: 15cfcd31d0561a60a7e7a5fb04d3bb3d4889d41a6372dcdf7a0cf13256a3b01e
|
| 20 |
+
specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 21 |
+
ranking_weights:
|
| 22 |
+
latent: 0.42
|
| 23 |
+
tempo: 0.18
|
| 24 |
+
chroma: 0.10
|
| 25 |
+
energy: 0.10
|
| 26 |
+
stereo: 0.10
|
| 27 |
+
spectral: 0.10
|
| 28 |
+
seam_penalty_weight: 0.10
|
| 29 |
+
creativity_bonus_weight: 0.10
|
| 30 |
+
selection_tiebreak: candidate_index
|
| 31 |
+
scoring_region: extension_only[529200:705600]
|
| 32 |
+
arbitrary_wav: false
|
| 33 |
+
native_tokenizer: false
|
| 34 |
+
semantic_style_transfer: false
|
| 35 |
+
historical_phase0_rng: false
|
| 36 |
+
masked_flow_cleanup: false
|
configs/continuation-tier-a.yaml
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.captured-continuation-config.v2
|
| 2 |
+
capability: fresh_same_process_captured_prompt_continuation
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
dtype: bfloat16
|
| 7 |
+
device: cuda
|
| 8 |
+
append_frames: 25
|
| 9 |
+
num_inference_steps: 30
|
| 10 |
+
equal_power_overlap_samples: 11008
|
| 11 |
+
require_exact_appended_frames: true
|
| 12 |
+
cases:
|
| 13 |
+
- name: short_captured_continuation
|
| 14 |
+
phase0_case_name: short_parity
|
| 15 |
+
source_run_id: b0ee35156dad797b7d120419782d5c1d8847f24ae8c5c3aa3d0bd2e1dd7baa6d
|
| 16 |
+
continue_from_cache: true
|
| 17 |
+
ar_seed: 7007
|
| 18 |
+
- name: multichunk_replay
|
| 19 |
+
phase0_case_name: multichunk_acceptance
|
| 20 |
+
source_run_id: 2224b993a75741fb7391b545cb8687c14a131aec4568925bec3dd565ca84a704
|
| 21 |
+
continue_from_cache: false
|
| 22 |
+
ar_seed: 7013
|
| 23 |
+
|
| 24 |
+
# Tier-A requires authenticated native token/fused/flow capture state.
|
| 25 |
+
# Arbitrary-WAV continuation remains BLOCKED.
|
configs/external-flow-encoder-interim-678-v1.yaml
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.external-flow-encoder-finetune.interim678.v1
|
| 2 |
+
capability: external_waveform_to_continuous_flow_latents_not_native_rvq
|
| 3 |
+
initial_flow_config_file_sha256: 4a2474a50dd1e9e93c15cb5cf792e344531809a868b8f1b7405846a6012b47df
|
| 4 |
+
initial_flow_config_semantic_digest: 7b1ab53def143eca51e37af6089b28596f814ecdd11fc12312f8d76f05f13db3
|
| 5 |
+
initial_encoder_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
|
| 6 |
+
data:
|
| 7 |
+
accepted_count: 678
|
| 8 |
+
train_count: 542
|
| 9 |
+
validation_count: 68
|
| 10 |
+
heldout_count: 68
|
| 11 |
+
user22_count: 22
|
| 12 |
+
split_seed: 73
|
| 13 |
+
sample_rate: 44100
|
| 14 |
+
channels: 2
|
| 15 |
+
crop_samples: 44032
|
| 16 |
+
end_guard_seconds: 5
|
| 17 |
+
tranche_label: interim_tranche_678_due_systemic_bot_auth
|
| 18 |
+
dataset_status: not_2k_final
|
| 19 |
+
original_target_count: 2000
|
| 20 |
+
manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 21 |
+
splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
| 22 |
+
corpus_semantic_digest: 6cd403e9cd25e61d062af6caa9c6b136b751c7f668f1af7d0972989963ac3452
|
| 23 |
+
user22_hash_manifest_sha256: 924804bffd3c7cbc5d13f438a1a9ce8024842bfc0898504bc084af35d564a41a
|
| 24 |
+
training:
|
| 25 |
+
steps: 12000
|
| 26 |
+
batch_size: 8
|
| 27 |
+
external_batch_size: 6
|
| 28 |
+
music3_batch_size: 2
|
| 29 |
+
seed: 7302026
|
| 30 |
+
warmup_steps: 500
|
| 31 |
+
maximum_learning_rate: 0.0001
|
| 32 |
+
minimum_learning_rate: 0.00001
|
| 33 |
+
beta1: 0.9
|
| 34 |
+
beta2: 0.999
|
| 35 |
+
epsilon: 0.00000001
|
| 36 |
+
weight_decay: 0.0001
|
| 37 |
+
gradient_clip_norm: 1.0
|
| 38 |
+
validation_interval_steps: 500
|
| 39 |
+
autocast_dtype: bfloat16
|
| 40 |
+
master_parameter_dtype: float32
|
| 41 |
+
loss:
|
| 42 |
+
external_ruler_weight: 0.70
|
| 43 |
+
teacher_weight: 0.30
|
| 44 |
+
teacher_latent_nmse_weight: 1.0
|
| 45 |
+
teacher_ruler_weight: 0.05
|
| 46 |
+
external_prior_weight: 0.005
|
| 47 |
+
prior_latent_mean: 0.07592402398586273
|
| 48 |
+
prior_latent_std: 2.206683874130249
|
| 49 |
+
prior_tail_threshold: 6.0
|
| 50 |
+
prior_tail_weight: 0.1
|
| 51 |
+
evaluation:
|
| 52 |
+
minimum_external_ruler_improvement_fraction: 0.10
|
| 53 |
+
minimum_external_si_sdr_improvement_db: 1.0
|
| 54 |
+
minimum_external_correlation_improvement: 0.03
|
| 55 |
+
maximum_teacher_latent_nmse_regression_fraction: 0.05
|
| 56 |
+
maximum_teacher_ruler_regression_fraction: 0.05
|
| 57 |
+
maximum_one_pass_latency_ms: 2.0
|
| 58 |
+
user22_report_after_checkpoint_lock: true
|
| 59 |
+
selection_split: validation
|
| 60 |
+
frozen_vocoder: true
|
| 61 |
+
user22_selection_forbidden: true
|
| 62 |
+
native_tokenizer: false
|
| 63 |
+
generalization_claim: false
|
configs/external-flow-encoder-v1.yaml
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.external-flow-encoder-finetune.v1
|
| 2 |
+
capability: external_waveform_to_continuous_flow_latents_not_native_rvq
|
| 3 |
+
initial_flow_config_file_sha256: 4a2474a50dd1e9e93c15cb5cf792e344531809a868b8f1b7405846a6012b47df
|
| 4 |
+
initial_flow_config_semantic_digest: 7b1ab53def143eca51e37af6089b28596f814ecdd11fc12312f8d76f05f13db3
|
| 5 |
+
initial_encoder_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
|
| 6 |
+
data:
|
| 7 |
+
accepted_count: 2000
|
| 8 |
+
train_count: 1600
|
| 9 |
+
validation_count: 200
|
| 10 |
+
heldout_count: 200
|
| 11 |
+
user22_count: 22
|
| 12 |
+
split_seed: 73
|
| 13 |
+
sample_rate: 44100
|
| 14 |
+
channels: 2
|
| 15 |
+
crop_samples: 44032
|
| 16 |
+
end_guard_seconds: 5
|
| 17 |
+
training:
|
| 18 |
+
steps: 12000
|
| 19 |
+
batch_size: 8
|
| 20 |
+
external_batch_size: 6
|
| 21 |
+
music3_batch_size: 2
|
| 22 |
+
seed: 7302026
|
| 23 |
+
warmup_steps: 500
|
| 24 |
+
maximum_learning_rate: 0.0001
|
| 25 |
+
minimum_learning_rate: 0.00001
|
| 26 |
+
beta1: 0.9
|
| 27 |
+
beta2: 0.999
|
| 28 |
+
epsilon: 0.00000001
|
| 29 |
+
weight_decay: 0.0001
|
| 30 |
+
gradient_clip_norm: 1.0
|
| 31 |
+
validation_interval_steps: 500
|
| 32 |
+
autocast_dtype: bfloat16
|
| 33 |
+
master_parameter_dtype: float32
|
| 34 |
+
loss:
|
| 35 |
+
external_ruler_weight: 0.70
|
| 36 |
+
teacher_weight: 0.30
|
| 37 |
+
teacher_latent_nmse_weight: 1.0
|
| 38 |
+
teacher_ruler_weight: 0.05
|
| 39 |
+
external_prior_weight: 0.005
|
| 40 |
+
prior_latent_mean: 0.07592402398586273
|
| 41 |
+
prior_latent_std: 2.206683874130249
|
| 42 |
+
prior_tail_threshold: 6.0
|
| 43 |
+
prior_tail_weight: 0.1
|
| 44 |
+
evaluation:
|
| 45 |
+
minimum_external_ruler_improvement_fraction: 0.10
|
| 46 |
+
minimum_external_si_sdr_improvement_db: 1.0
|
| 47 |
+
minimum_external_correlation_improvement: 0.03
|
| 48 |
+
maximum_teacher_latent_nmse_regression_fraction: 0.05
|
| 49 |
+
maximum_teacher_ruler_regression_fraction: 0.05
|
| 50 |
+
maximum_one_pass_latency_ms: 2.0
|
| 51 |
+
user22_report_after_checkpoint_lock: true
|
| 52 |
+
selection_split: validation
|
| 53 |
+
frozen_vocoder: true
|
| 54 |
+
user22_selection_forbidden: true
|
| 55 |
+
native_tokenizer: false
|
| 56 |
+
generalization_claim: false
|
configs/flow-encoder-v1.yaml
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.flow-encoder-config.v1
|
| 2 |
+
capability: waveform_to_continuous_flow_vocoder_latents_not_native_rvq
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
teacher:
|
| 7 |
+
prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
|
| 8 |
+
lyrics: "[instrumental]"
|
| 9 |
+
duration_seconds: 1.0
|
| 10 |
+
inference_steps: 12
|
| 11 |
+
guidance_scale: 1.7
|
| 12 |
+
train_seed_start: 1000
|
| 13 |
+
train_count: 64
|
| 14 |
+
validation_seed_start: 2000
|
| 15 |
+
validation_count: 16
|
| 16 |
+
heldout_seed_start: 3000
|
| 17 |
+
heldout_count: 16
|
| 18 |
+
model:
|
| 19 |
+
sample_rate: 44100
|
| 20 |
+
input_channels: 2
|
| 21 |
+
input_samples: 44032
|
| 22 |
+
latent_channels: 128
|
| 23 |
+
latent_frames: 86
|
| 24 |
+
frontend_kernel: 1024
|
| 25 |
+
frontend_stride: 512
|
| 26 |
+
frontend_padding: 256
|
| 27 |
+
width: 192
|
| 28 |
+
group_norm_groups: 24
|
| 29 |
+
residual_dilations: [1, 3, 9, 27]
|
| 30 |
+
training:
|
| 31 |
+
steps: 2000
|
| 32 |
+
batch_size: 8
|
| 33 |
+
seed: 424242
|
| 34 |
+
maximum_learning_rate: 0.0003
|
| 35 |
+
minimum_learning_rate: 0.00003
|
| 36 |
+
beta1: 0.9
|
| 37 |
+
beta2: 0.999
|
| 38 |
+
epsilon: 0.00000001
|
| 39 |
+
weight_decay: 0.0001
|
| 40 |
+
gradient_clip_norm: 1.0
|
| 41 |
+
validation_interval_steps: 100
|
| 42 |
+
latent_loss_weight: 1.0
|
| 43 |
+
audio_ruler_weight: 0.05
|
| 44 |
+
master_parameter_dtype: float32
|
| 45 |
+
autocast_dtype: bfloat16
|
| 46 |
+
decoder_dtype: bfloat16
|
| 47 |
+
loss:
|
| 48 |
+
latent_mean: 0.07592402398586273
|
| 49 |
+
latent_std: 2.206683874130249
|
| 50 |
+
stft_center: false
|
| 51 |
+
stft_fft_sizes: [256, 512, 1024, 2048]
|
| 52 |
+
stft_hop_sizes: [64, 128, 256, 512]
|
| 53 |
+
time_denominator_epsilon: 0.00000001
|
| 54 |
+
complex_stft_denominator_epsilon: 0.00000001
|
| 55 |
+
legacy_mrstft_epsilon: 0.0000001
|
| 56 |
+
mid_side_floor_fraction: 0.0001
|
| 57 |
+
envelope_windows: [63, 255, 1023]
|
| 58 |
+
relative_envelope_denominator_epsilon: 0.00000001
|
| 59 |
+
ruler_weights:
|
| 60 |
+
time_nmse: 2.0
|
| 61 |
+
complex_stft_nmse: 1.0
|
| 62 |
+
legacy_mrstft: 0.05
|
| 63 |
+
mid_side_nmse: 0.5
|
| 64 |
+
relative_envelope: 0.02
|
| 65 |
+
evaluation:
|
| 66 |
+
refinement_steps: 20
|
| 67 |
+
refinement_learning_rate: 0.01
|
| 68 |
+
minimum_holdout_latent_mse_improvement_fraction: 0.05
|
| 69 |
+
minimum_holdout_audio_ruler_improvement_fraction: 0.02
|
| 70 |
+
external_crop_sha256: 1c3d61f242412a78396665da14c8bcb4002dc93f5518fe81c0303fe9c0b77477
|
| 71 |
+
external_source_id: f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
|
| 72 |
+
publication:
|
| 73 |
+
private_dataset_root_mode: 448
|
| 74 |
+
private_dataset_file_mode: 384
|
| 75 |
+
evidence_root_mode: 493
|
| 76 |
+
evidence_file_mode: 420
|
| 77 |
+
manifest_written_last: true
|
| 78 |
+
atomic_directory_rename: true
|
configs/flow-inpaint-v1.yaml
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.masked-flow-inpaint-config.v1
|
| 2 |
+
capability: captured_condition_flow_latent_reconstruction_only
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
teacher:
|
| 7 |
+
prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
|
| 8 |
+
lyrics: "[instrumental]"
|
| 9 |
+
duration_seconds: 1.0
|
| 10 |
+
inference_steps: 12
|
| 11 |
+
guidance_scale: 1.7
|
| 12 |
+
train_seed_start: 1000
|
| 13 |
+
train_count: 64
|
| 14 |
+
validation_seed_start: 2000
|
| 15 |
+
validation_count: 16
|
| 16 |
+
heldout_seed_start: 3000
|
| 17 |
+
heldout_count: 16
|
| 18 |
+
adapter:
|
| 19 |
+
latent_channels: 128
|
| 20 |
+
frames: 86
|
| 21 |
+
condition_channels: 2048
|
| 22 |
+
flow_layers: 36
|
| 23 |
+
flow_hidden_size: 2048
|
| 24 |
+
lora_rank: 4
|
| 25 |
+
lora_scale: 1.0
|
| 26 |
+
region_embedding_count: 3
|
| 27 |
+
expected_trainable_parameters: 1775616
|
| 28 |
+
hole:
|
| 29 |
+
minimum_frames: 16
|
| 30 |
+
maximum_frames: 32
|
| 31 |
+
minimum_left_context_frames: 16
|
| 32 |
+
minimum_right_context_frames: 16
|
| 33 |
+
frame_samples: 512
|
| 34 |
+
training:
|
| 35 |
+
steps: 2000
|
| 36 |
+
batch_size: 8
|
| 37 |
+
seed: 731942
|
| 38 |
+
maximum_learning_rate: 0.0002
|
| 39 |
+
minimum_learning_rate: 0.00002
|
| 40 |
+
beta1: 0.9
|
| 41 |
+
beta2: 0.999
|
| 42 |
+
epsilon: 0.00000001
|
| 43 |
+
weight_decay: 0.0
|
| 44 |
+
gradient_clip_norm: 1.0
|
| 45 |
+
validation_interval_steps: 200
|
| 46 |
+
gradient_checkpointing: true
|
| 47 |
+
parameter_dtype: float32
|
| 48 |
+
compute_dtype: bfloat16
|
| 49 |
+
inference:
|
| 50 |
+
euler_steps: 30
|
| 51 |
+
guidance_scale: 1.7
|
| 52 |
+
evaluation_mask_seed: 880301
|
| 53 |
+
evaluation:
|
| 54 |
+
latent_std: 2.206683874130249
|
| 55 |
+
minimum_median_hole_latent_nmse_improvement_fraction: 0.05
|
| 56 |
+
minimum_median_hole_audio_ruler_improvement_fraction: 0.05
|
| 57 |
+
audio_ruler_config_path: flow-encoder-v1.yaml
|
| 58 |
+
publication:
|
| 59 |
+
private_dataset_root_mode: 448
|
| 60 |
+
private_dataset_file_mode: 384
|
| 61 |
+
evidence_root_mode: 493
|
| 62 |
+
evidence_file_mode: 420
|
| 63 |
+
manifest_written_last: true
|
| 64 |
+
atomic_directory_rename: true
|
| 65 |
+
|
| 66 |
+
# This pilot reconstructs masked regions of captured Music3 Flow state only.
|
| 67 |
+
# It does not accept arbitrary WAV, expose native RVQ tokens, or prove FIM/generalization.
|
configs/flow-prepend-v1.yaml
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.flow-prepend-config.v1
|
| 2 |
+
capability: local_one_second_captured_condition_renderer_latent_prefix_reconstruction_only
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
dataset:
|
| 7 |
+
schema_version: music3lab.flow-encoder-teachers.v2
|
| 8 |
+
semantic_digest: 0b8fb727a380b5a57373f12fb4ffd31eca65c2bdf855d4eee404c2380b3e5607
|
| 9 |
+
source_config_file_sha256: 3aaf6d48f755ce13ca031e2326c7caf830a53fba2054f827099317f4efe436b8
|
| 10 |
+
source_config_semantic_digest: de5f75e471772ac1aeb9805fce87c3293135f96fd21936e058cef8afa97f62c5
|
| 11 |
+
train_count: 64
|
| 12 |
+
validation_count: 16
|
| 13 |
+
heldout_count: 16
|
| 14 |
+
adapter:
|
| 15 |
+
latent_channels: 128
|
| 16 |
+
frames: 86
|
| 17 |
+
condition_channels: 2048
|
| 18 |
+
flow_layers: 36
|
| 19 |
+
flow_hidden_size: 2048
|
| 20 |
+
lora_rank: 4
|
| 21 |
+
lora_scale: 1.0
|
| 22 |
+
region_embedding_count: 4
|
| 23 |
+
expected_trainable_parameters: 1777664
|
| 24 |
+
masks:
|
| 25 |
+
minimum_frames: 16
|
| 26 |
+
maximum_frames: 32
|
| 27 |
+
minimum_left_context_frames: 16
|
| 28 |
+
minimum_right_context_frames: 16
|
| 29 |
+
two_sided_fraction: 0.5
|
| 30 |
+
prefix_fraction: 0.5
|
| 31 |
+
frame_samples: 512
|
| 32 |
+
training:
|
| 33 |
+
steps: 2000
|
| 34 |
+
batch_size: 8
|
| 35 |
+
seed: 731942
|
| 36 |
+
maximum_learning_rate: 0.0002
|
| 37 |
+
minimum_learning_rate: 0.00002
|
| 38 |
+
beta1: 0.9
|
| 39 |
+
beta2: 0.999
|
| 40 |
+
epsilon: 0.00000001
|
| 41 |
+
weight_decay: 0.0
|
| 42 |
+
gradient_clip_norm: 1.0
|
| 43 |
+
validation_interval_steps: 200
|
| 44 |
+
gradient_checkpointing: true
|
| 45 |
+
parameter_dtype: float32
|
| 46 |
+
compute_dtype: bfloat16
|
| 47 |
+
inference:
|
| 48 |
+
euler_steps: 30
|
| 49 |
+
guidance_scale: 1.7
|
| 50 |
+
evaluation_mask_seed: 880301
|
| 51 |
+
evaluation:
|
| 52 |
+
latent_std: 2.206683874130249
|
| 53 |
+
minimum_median_prefix_latent_nmse_improvement_fraction: 0.05
|
| 54 |
+
minimum_median_prefix_audio_ruler_improvement_fraction: 0.05
|
| 55 |
+
audio_ruler_config_path: flow-encoder-v1.yaml
|
| 56 |
+
publication:
|
| 57 |
+
private_dataset_root_mode: 448
|
| 58 |
+
private_dataset_file_mode: 384
|
| 59 |
+
evidence_root_mode: 493
|
| 60 |
+
evidence_file_mode: 420
|
| 61 |
+
manifest_written_last: true
|
| 62 |
+
atomic_directory_rename: true
|
| 63 |
+
|
| 64 |
+
# This reuses authenticated one-second captured Flow state and evaluates only a
|
| 65 |
+
# local renderer-latent prefix. It is not song-intro/native-AR prepend,
|
| 66 |
+
# arbitrary-WAV editing, FIM, native-token inference, or generalization.
|
configs/inversion-e3.yaml
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.inversion-e3-config.v1
|
| 2 |
+
experiment_label: external_audio_latent_inversion_no_oracle
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
external_source:
|
| 7 |
+
kind: opaque_user_audio
|
| 8 |
+
raw_file_sha256: f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
|
| 9 |
+
raw_file_size: 31437094
|
| 10 |
+
container: wav
|
| 11 |
+
codec: pcm_s32le
|
| 12 |
+
source_sample_rate: 48000
|
| 13 |
+
source_channels: 2
|
| 14 |
+
source_frames: 3929624
|
| 15 |
+
retain_original_bytes: true
|
| 16 |
+
persist_human_metadata: false
|
| 17 |
+
canonicalization:
|
| 18 |
+
engine: ffmpeg
|
| 19 |
+
executable: /usr/bin/ffmpeg
|
| 20 |
+
executable_sha256: 36d94a605d612e4090d1b8aec889d0c0801c6eafb1593c90f5c0dfd2e2966a45
|
| 21 |
+
version_line: ffmpeg version 4.4.2-0ubuntu0.22.04.1 Copyright (c) 2000-2021 the FFmpeg developers
|
| 22 |
+
input_transport: retained_source_bytes_via_stdin
|
| 23 |
+
operation_order: resample_full_source_then_trim_output_frames
|
| 24 |
+
filtergraph: aresample=44100,atrim=start_sample=1719900:end_sample=1763932,asetpts=PTS-STARTPTS
|
| 25 |
+
output_encoding: stereo_interleaved_little_endian_float32
|
| 26 |
+
output_sample_rate: 44100
|
| 27 |
+
output_channels: 2
|
| 28 |
+
output_frames: 44032
|
| 29 |
+
output_start_frame: 1719900
|
| 30 |
+
output_end_frame: 1763932
|
| 31 |
+
interleaved_f32le_sha256: 22d81730b21386f516bbb586f3310acb7aaed75b157a52dca9f0c7c76ef4b3bd
|
| 32 |
+
channel_major_tensor_sha256: b08e725241d7e3158e1a286ba9279e903b1a6a2f3f6c749af276ff02be0017b1
|
| 33 |
+
deterministic_float_wav_sha256: 2645ff4f920bb559cd8e4898f91e06c0f43bbfda715ccab2f08da1477440f3b7
|
| 34 |
+
execution:
|
| 35 |
+
device: cuda
|
| 36 |
+
required_device_name_substring: H100
|
| 37 |
+
deterministic_algorithms: true
|
| 38 |
+
restart_seeds: [101, 103, 107, 109]
|
| 39 |
+
master_latent_dtype: float32
|
| 40 |
+
fp32_decoder_dtype: float32
|
| 41 |
+
exact_decoder_dtype: bfloat16
|
| 42 |
+
trajectory_stride_steps: 1
|
| 43 |
+
latent_channels: 128
|
| 44 |
+
latent_frames: 86
|
| 45 |
+
latent_initialization_mean: 0.07592402398586273
|
| 46 |
+
latent_initialization_std: 2.206683874130249
|
| 47 |
+
initial_batch_sha256: d907a81862d25e1bf3bbf54b78b801708f723ba6b654759bbee813c304aec2f1
|
| 48 |
+
initial_restart_sha256:
|
| 49 |
+
- 6832687067d4c42e6b2f634c5e1648d12418259064c1fc71acb792e1c2ebefad
|
| 50 |
+
- eaa511076d9185d4f95dfc6f3aa6feae56f2002ecb8606c338dc1e12ce26e700
|
| 51 |
+
- 73ef18c7dda783bb61f0cfe4a4e0d71a5e17f1bfdd9455de6054e77228c7fdbb
|
| 52 |
+
- bcd53046e276188aee1bf4af4bfd8dddaf07513f67eaa7893158d055ce1dc4e5
|
| 53 |
+
optimizer:
|
| 54 |
+
name: adam
|
| 55 |
+
beta1: 0.9
|
| 56 |
+
beta2: 0.999
|
| 57 |
+
epsilon: 0.00000001
|
| 58 |
+
weight_decay: 0.0
|
| 59 |
+
gradient_clip_norm: 1.0
|
| 60 |
+
warmup_steps: 20
|
| 61 |
+
schedule: linear_warmup_then_cosine_inclusive
|
| 62 |
+
loss:
|
| 63 |
+
definition_version: fixed-bf16-ruler-v1
|
| 64 |
+
stft_center: false
|
| 65 |
+
stft_fft_sizes: [256, 512, 1024, 2048]
|
| 66 |
+
stft_hop_sizes: [64, 128, 256, 512]
|
| 67 |
+
legacy_mrstft_epsilon: 0.0000001
|
| 68 |
+
envelope_windows: [63, 255, 1023]
|
| 69 |
+
time_denominator_epsilon: 0.00000001
|
| 70 |
+
complex_stft_denominator_epsilon: 0.00000001
|
| 71 |
+
mid_side_floor_fraction: 0.0001
|
| 72 |
+
relative_envelope_denominator_epsilon: 0.00000001
|
| 73 |
+
fp32_audio_weights_start:
|
| 74 |
+
time_nmse: 1.0
|
| 75 |
+
complex_stft_nmse: 0.25
|
| 76 |
+
legacy_mrstft: 0.5
|
| 77 |
+
mid_side_nmse: 0.1
|
| 78 |
+
relative_envelope: 0.1
|
| 79 |
+
fixed_selection_weights:
|
| 80 |
+
time_nmse: 2.0
|
| 81 |
+
complex_stft_nmse: 1.0
|
| 82 |
+
legacy_mrstft: 0.05
|
| 83 |
+
mid_side_nmse: 0.5
|
| 84 |
+
relative_envelope: 0.02
|
| 85 |
+
fp32_prior_start: 0.02
|
| 86 |
+
fp32_prior_end: 0.005
|
| 87 |
+
bf16_prior_start: 0.005
|
| 88 |
+
bf16_prior_end: 0.001
|
| 89 |
+
distribution_prior:
|
| 90 |
+
definition: population_moment_and_tail_v1
|
| 91 |
+
latent_mean: 0.07592402398586273
|
| 92 |
+
latent_std: 2.206683874130249
|
| 93 |
+
variance_epsilon: 0.00000001
|
| 94 |
+
tail_threshold: 6.0
|
| 95 |
+
tail_weight: 0.1
|
| 96 |
+
formula: prior=mean(z)^2+(sqrt(mean((z-mean(z))^2)+1e-8)-1)^2+0.1*mean(relu(abs(z)-6)^2)
|
| 97 |
+
oracle_distance_used: false
|
| 98 |
+
selection:
|
| 99 |
+
score: fixed_exact_bf16_audio_ruler
|
| 100 |
+
evaluate_every_optimizer_step: true
|
| 101 |
+
per_restart_best: true
|
| 102 |
+
stage_transition_uses_fp32_stage_best: true
|
| 103 |
+
restart_tie_rule: earliest_stage_then_step
|
| 104 |
+
final_tie_rule: lowest_restart_index
|
| 105 |
+
evaluator_computed_after_lock: true
|
| 106 |
+
prior_excluded: true
|
| 107 |
+
experiment:
|
| 108 |
+
experiment_id: P2-E3
|
| 109 |
+
kind: external_audio_inversion
|
| 110 |
+
initialization: four_independent_seeded_prior_draws
|
| 111 |
+
fp32_stage:
|
| 112 |
+
steps: 800
|
| 113 |
+
maximum_learning_rate: 0.03
|
| 114 |
+
minimum_learning_rate: 0.003
|
| 115 |
+
bf16_stage:
|
| 116 |
+
steps: 1200
|
| 117 |
+
maximum_learning_rate: 0.015
|
| 118 |
+
minimum_learning_rate: 0.0015
|
| 119 |
+
audio_only_thresholds:
|
| 120 |
+
minimum_objective_improvement_fraction: 0.20
|
| 121 |
+
minimum_median_objective_improvement_fraction: 0.40
|
| 122 |
+
maximum_waveform_mae: 0.06
|
| 123 |
+
minimum_correlation: 0.50
|
| 124 |
+
minimum_si_sdr_db: -1.0
|
| 125 |
+
high_fidelity_minimum_si_sdr_db: 20.0
|
| 126 |
+
high_fidelity_minimum_unscaled_snr_db: 18.0
|
| 127 |
+
high_fidelity_maximum_loudness_error_db: 0.5
|
| 128 |
+
high_fidelity_maximum_stereo_correlation_error: 0.05
|
| 129 |
+
publication:
|
| 130 |
+
root_mode: 493
|
| 131 |
+
experiment_directory_mode: 448
|
| 132 |
+
file_mode: 420
|
| 133 |
+
manifest_written_last: true
|
| 134 |
+
atomic_directory_rename: true
|
| 135 |
+
notes:
|
| 136 |
+
capability: continuous Flow-VAE latent inversion from external waveform; not WAV-to-RVQ encoding
|
| 137 |
+
latent_quality_claim_allowed: false
|
| 138 |
+
human_metadata_persisted: false
|
| 139 |
+
quality_claim_policy: audio-only feasibility and high-fidelity tiers are separate
|
configs/inversion-v1.yaml
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.inversion-config.v1.2
|
| 2 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 3 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 4 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 5 |
+
oracle:
|
| 6 |
+
kind: short
|
| 7 |
+
run_id: 2a1bc3404f7a0c1c7bc03fc3813df5e8a3769aa9b413121f1669edd097353368
|
| 8 |
+
sampling_rate: 44100
|
| 9 |
+
channels: 2
|
| 10 |
+
samples: 44032
|
| 11 |
+
latent_channels: 128
|
| 12 |
+
latent_frames: 86
|
| 13 |
+
execution:
|
| 14 |
+
device: cuda
|
| 15 |
+
required_device_name_substring: H100
|
| 16 |
+
vocoder_dtype: bfloat16
|
| 17 |
+
master_latent_dtype: float32
|
| 18 |
+
deterministic_algorithms: true
|
| 19 |
+
restart_seeds: [101, 103, 107, 109]
|
| 20 |
+
trace_interval_steps: 10
|
| 21 |
+
loss:
|
| 22 |
+
charbonnier_epsilon: 0.001
|
| 23 |
+
stft_epsilon: 0.0000001
|
| 24 |
+
stft_center: false
|
| 25 |
+
definition_version: mrstft-center-false-unscaled-snr-v1
|
| 26 |
+
stft_fft_sizes: [256, 512, 1024, 2048]
|
| 27 |
+
stft_hop_sizes: [64, 128, 256, 512]
|
| 28 |
+
envelope_windows: [63, 255, 1023]
|
| 29 |
+
weights:
|
| 30 |
+
waveform_charbonnier: 1.0
|
| 31 |
+
mrstft: 0.15
|
| 32 |
+
mid_side_charbonnier: 0.25
|
| 33 |
+
multiscale_envelope: 0.15
|
| 34 |
+
latent_prior: 0.000001
|
| 35 |
+
experiments:
|
| 36 |
+
- experiment_id: P2-E1
|
| 37 |
+
kind: perturbed_oracle_recovery
|
| 38 |
+
steps: 240
|
| 39 |
+
learning_rate: 0.05
|
| 40 |
+
gradient_clip_norm: 2.0
|
| 41 |
+
initialization:
|
| 42 |
+
distribution: perturbed_oracle
|
| 43 |
+
mean: 0.0
|
| 44 |
+
std: 0.08
|
| 45 |
+
thresholds:
|
| 46 |
+
minimum_objective_improvement_fraction: 0.75
|
| 47 |
+
minimum_median_objective_improvement_fraction: 0.40
|
| 48 |
+
maximum_waveform_mae: 0.008
|
| 49 |
+
minimum_correlation: 0.99
|
| 50 |
+
minimum_si_sdr_db: 22.0
|
| 51 |
+
maximum_latent_rmse: 0.08
|
| 52 |
+
high_fidelity_minimum_si_sdr_db: 20.0
|
| 53 |
+
high_fidelity_minimum_unscaled_snr_db: 18.0
|
| 54 |
+
high_fidelity_maximum_loudness_error_db: 0.5
|
| 55 |
+
high_fidelity_maximum_stereo_correlation_error: 0.05
|
| 56 |
+
- experiment_id: P2-E2
|
| 57 |
+
kind: generated_waveform_inversion
|
| 58 |
+
steps: 240
|
| 59 |
+
learning_rate: 0.08
|
| 60 |
+
gradient_clip_norm: 2.0
|
| 61 |
+
initialization:
|
| 62 |
+
distribution: random_normal
|
| 63 |
+
mean: 0.075924
|
| 64 |
+
std: 2.206684
|
| 65 |
+
thresholds:
|
| 66 |
+
minimum_objective_improvement_fraction: 0.20
|
| 67 |
+
minimum_median_objective_improvement_fraction: 0.40
|
| 68 |
+
maximum_waveform_mae: 0.06
|
| 69 |
+
minimum_correlation: 0.50
|
| 70 |
+
minimum_si_sdr_db: -1.0
|
| 71 |
+
maximum_latent_rmse: null
|
| 72 |
+
high_fidelity_minimum_si_sdr_db: 20.0
|
| 73 |
+
high_fidelity_minimum_unscaled_snr_db: 18.0
|
| 74 |
+
high_fidelity_maximum_loudness_error_db: 0.5
|
| 75 |
+
high_fidelity_maximum_stereo_correlation_error: 0.05
|
| 76 |
+
selection:
|
| 77 |
+
criterion: final_optimization_objective
|
| 78 |
+
evaluator_metrics_computed_after_selection: true
|
| 79 |
+
permit_evaluator_metric_selection: false
|
| 80 |
+
tie_rule: lowest_restart_index
|
| 81 |
+
publication:
|
| 82 |
+
root_mode: 493
|
| 83 |
+
experiment_directory_mode: 448
|
| 84 |
+
file_mode: 420
|
| 85 |
+
manifest_written_last: true
|
| 86 |
+
atomic_directory_rename: true
|
| 87 |
+
notes:
|
| 88 |
+
target_provenance: >-
|
| 89 |
+
Exact BF16-generated Phase-0 short_parity waveform and final continuous
|
| 90 |
+
Flow-VAE latent oracle; this experiment does not encode a WAV into native
|
| 91 |
+
RVQ tokens.
|
| 92 |
+
quality_claim_policy: >-
|
| 93 |
+
Report no inversion-quality success unless every preregistered threshold
|
| 94 |
+
for that experiment passes.
|
configs/inversion-v2.yaml
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.inversion-v2-config.v2
|
| 2 |
+
experiment_label: latent_only_continuation_not_optimizer_resume
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
authority_migration:
|
| 7 |
+
kind: v1.1_to_v1.3_authority_only
|
| 8 |
+
preregistered_config_sha256: dba79de7191ed4a4242f4fecb691314b04fde945a1eb9f4e1fa1c97a6a017a3f
|
| 9 |
+
preserved_sections_sha256: 287b6aed4d19ec63f3f78bd1ee414eb3a6c1f56b7f6dc961fb5e0ea4c02ff7a1
|
| 10 |
+
only_changed_top_level_keys: [schema_version, v1_authority, authority_migration]
|
| 11 |
+
v1_authority:
|
| 12 |
+
schema_version: music3lab.inversion-session.v1.3
|
| 13 |
+
root_name: inversion-v1.3-3632094
|
| 14 |
+
session_file_sha256: bf615c719ddaa53fd300512867694447ea91c6703fa709f250b28484ecf68445
|
| 15 |
+
session_semantic_digest: 1554159fc7a5e364de063e3c8345c3cf05426288aa23bd0039957026aabc3a7c
|
| 16 |
+
config_file_sha256: e79362ec680b6e2af163c17da8942675eb31d4084aa301abd143646cf0d51779
|
| 17 |
+
config_semantic_digest: 09ee76c6aeec8a91733391b386f1c8f043df5a9919beb9acb29b718199e2900c
|
| 18 |
+
project_git_commit: 36320944c624faefba78459ae07f9e0d1faea874
|
| 19 |
+
project_source_sha256: 9b2a741ed6ee5ff9763a2fb81aee0194d37d469c8e676c213ddf5aa9418a1921
|
| 20 |
+
adapter_semantic_digest: 2c7cd7a868d1bd3f0ea670c44bd9a1b806b58fbb4b2293264b04a279e7d285ea
|
| 21 |
+
oracle_semantic_digest: b1f4ed241802c3d52c1a8fef8ca882177f2f3342d07e586e50ffa0f095609347
|
| 22 |
+
p2e1_manifest_file_sha256: 18503958f437f3c7aef5a807bac1f26d7822acf55f5c5c8813ddfcc8d2a82987
|
| 23 |
+
p2e1_manifest_semantic_digest: 2dc5af49efa39a313519ea908a6a4694c98663b780c79f99caf8272271d5e161
|
| 24 |
+
p2e2_manifest_file_sha256: b27a1093a53673ba8f10eb40f6afbc7ae1ad43728037ea306a2ba59a7a7e1760
|
| 25 |
+
p2e2_manifest_semantic_digest: ed9fa2c63aa6c8c8f41b0fb3adab474f9ce1ab8cb803d62d945f8db4e2249b6e
|
| 26 |
+
p2e1_latents_file_sha256: dd56e055c39d7344d501daa9ae47c4c4da07c583278751bf4b1ccc4e747f0e57
|
| 27 |
+
p2e2_latents_file_sha256: 8e4e974adb69e00911bb041c957398db7dd07317fe6120550d31929781d2a571
|
| 28 |
+
p2e1_final_latents_sha256: e687c11b2eb406821e8b972d1edcd5a956af587b9787230fcda089d23689d336
|
| 29 |
+
p2e1_final_restart_sha256:
|
| 30 |
+
- 6f0276e1d6139570ba8789bdb84a7463cdbcdebcc030473e1e314693b658bb73
|
| 31 |
+
- 5db91d5d300a9d3bd5aa171113cf7b6849075bb00a0be23dc279d19e7fb28c53
|
| 32 |
+
- 996cb44f54f01c5a0495bb4390b7a5322023f52eba28b3f20729bb669dba62d3
|
| 33 |
+
- 92fa103dee0a71a787e9c5037d935104032e665e783e231886c45a6b6ea1dc83
|
| 34 |
+
p2e2_final_latents_sha256: 6287ffa20f336188e2e24c9bdd0f992518127c7eefdb05123adcf62e36d100c6
|
| 35 |
+
p2e2_final_restart_sha256:
|
| 36 |
+
- 518f00fa4b22a2a334efca0efafd669f6d9e39b092a84934bdf440753ca680b7
|
| 37 |
+
- 2a25806f5065c4f89e796f135e8162af60b1e781209c923bbc6f134deac020aa
|
| 38 |
+
- e0a59a93c125502fa104ecd5efc2a619c10fc0c5a8efc9d65991ab747659edac
|
| 39 |
+
- 144aafa55d311a8d05eb01566d61d7b223d13ca8d06347760ee0b237751782c3
|
| 40 |
+
p2e1_seeded_restart_sha256:
|
| 41 |
+
- bc8b5700a710bad531dd41cea022a3e74d317c174027e6efa5144acc02c443fe
|
| 42 |
+
- 52f2d55b630cb86ae26cf6f7ef656c73b1e0bf2fbec4023e2f3e91cfba5e3d9d
|
| 43 |
+
- 6b5ce53cdbcf27f594c5e5a68434af4f6f3d3dca6bfa1db2a9e75f083d32b831
|
| 44 |
+
- cae9150a6e73da5c0cec5196cb3390cb78ca956ebef1cbcacfa5dc069788cd51
|
| 45 |
+
oracle:
|
| 46 |
+
kind: short
|
| 47 |
+
run_id: 2a1bc3404f7a0c1c7bc03fc3813df5e8a3769aa9b413121f1669edd097353368
|
| 48 |
+
sampling_rate: 44100
|
| 49 |
+
channels: 2
|
| 50 |
+
samples: 44032
|
| 51 |
+
latent_channels: 128
|
| 52 |
+
latent_frames: 86
|
| 53 |
+
execution:
|
| 54 |
+
device: cuda
|
| 55 |
+
required_device_name_substring: H100
|
| 56 |
+
deterministic_algorithms: true
|
| 57 |
+
restart_seeds: [101, 103, 107, 109]
|
| 58 |
+
master_latent_dtype: float32
|
| 59 |
+
fp32_decoder_dtype: float32
|
| 60 |
+
exact_decoder_dtype: bfloat16
|
| 61 |
+
trajectory_stride_steps: 1
|
| 62 |
+
optimizer:
|
| 63 |
+
name: adam
|
| 64 |
+
beta1: 0.9
|
| 65 |
+
beta2: 0.999
|
| 66 |
+
epsilon: 0.00000001
|
| 67 |
+
weight_decay: 0.0
|
| 68 |
+
gradient_clip_norm: 1.0
|
| 69 |
+
warmup_steps: 20
|
| 70 |
+
schedule: linear_warmup_then_cosine_inclusive
|
| 71 |
+
loss:
|
| 72 |
+
definition_version: fixed-bf16-ruler-v1
|
| 73 |
+
stft_center: false
|
| 74 |
+
stft_fft_sizes: [256, 512, 1024, 2048]
|
| 75 |
+
stft_hop_sizes: [64, 128, 256, 512]
|
| 76 |
+
legacy_mrstft_epsilon: 0.0000001
|
| 77 |
+
envelope_windows: [63, 255, 1023]
|
| 78 |
+
time_denominator_epsilon: 0.00000001
|
| 79 |
+
complex_stft_denominator_epsilon: 0.00000001
|
| 80 |
+
mid_side_floor_fraction: 0.0001
|
| 81 |
+
relative_envelope_denominator_epsilon: 0.00000001
|
| 82 |
+
fp32_audio_weights_start:
|
| 83 |
+
time_nmse: 1.0
|
| 84 |
+
complex_stft_nmse: 0.25
|
| 85 |
+
legacy_mrstft: 0.5
|
| 86 |
+
mid_side_nmse: 0.1
|
| 87 |
+
relative_envelope: 0.1
|
| 88 |
+
fixed_selection_weights:
|
| 89 |
+
time_nmse: 2.0
|
| 90 |
+
complex_stft_nmse: 1.0
|
| 91 |
+
legacy_mrstft: 0.05
|
| 92 |
+
mid_side_nmse: 0.5
|
| 93 |
+
relative_envelope: 0.02
|
| 94 |
+
fp32_prior_start: 0.02
|
| 95 |
+
fp32_prior_end: 0.005
|
| 96 |
+
bf16_prior_start: 0.005
|
| 97 |
+
bf16_prior_end: 0.001
|
| 98 |
+
distribution_prior:
|
| 99 |
+
definition: population_moment_and_tail_v1
|
| 100 |
+
latent_mean: 0.07592402398586273
|
| 101 |
+
latent_std: 2.206683874130249
|
| 102 |
+
variance_epsilon: 0.00000001
|
| 103 |
+
tail_threshold: 6.0
|
| 104 |
+
tail_weight: 0.1
|
| 105 |
+
formula: prior=mean(z)^2+(sqrt(mean((z-mean(z))^2)+1e-8)-1)^2+0.1*mean(relu(abs(z)-6)^2)
|
| 106 |
+
oracle_distance_used: false
|
| 107 |
+
selection:
|
| 108 |
+
score: fixed_exact_bf16_audio_ruler
|
| 109 |
+
evaluate_every_optimizer_step: true
|
| 110 |
+
per_restart_best: true
|
| 111 |
+
stage_transition_uses_fp32_stage_best: true
|
| 112 |
+
restart_tie_rule: earliest_stage_then_step
|
| 113 |
+
final_tie_rule: lowest_restart_index
|
| 114 |
+
evaluator_computed_after_lock: true
|
| 115 |
+
prior_excluded: true
|
| 116 |
+
engineering_progress:
|
| 117 |
+
label: engineering_progress_3db_v1
|
| 118 |
+
definition: 10*log10(v1_fixed_bf16_score/v2_fixed_bf16_score)
|
| 119 |
+
minimum_db: 3.0
|
| 120 |
+
affects_quality_claim: false
|
| 121 |
+
experiments:
|
| 122 |
+
- experiment_id: P2-E1
|
| 123 |
+
kind: perturbed_oracle_recovery
|
| 124 |
+
initialization: original_seeded_perturbations_not_v1_optimizer_state
|
| 125 |
+
fp32_stage:
|
| 126 |
+
steps: 600
|
| 127 |
+
maximum_learning_rate: 0.02
|
| 128 |
+
minimum_learning_rate: 0.002
|
| 129 |
+
bf16_stage:
|
| 130 |
+
steps: 600
|
| 131 |
+
maximum_learning_rate: 0.008
|
| 132 |
+
minimum_learning_rate: 0.0008
|
| 133 |
+
thresholds:
|
| 134 |
+
minimum_objective_improvement_fraction: 0.75
|
| 135 |
+
minimum_median_objective_improvement_fraction: 0.40
|
| 136 |
+
maximum_waveform_mae: 0.008
|
| 137 |
+
minimum_correlation: 0.99
|
| 138 |
+
minimum_si_sdr_db: 22.0
|
| 139 |
+
maximum_latent_rmse: 0.08
|
| 140 |
+
high_fidelity_minimum_si_sdr_db: 20.0
|
| 141 |
+
high_fidelity_minimum_unscaled_snr_db: 18.0
|
| 142 |
+
high_fidelity_maximum_loudness_error_db: 0.5
|
| 143 |
+
high_fidelity_maximum_stereo_correlation_error: 0.05
|
| 144 |
+
- experiment_id: P2-E2
|
| 145 |
+
kind: generated_waveform_inversion
|
| 146 |
+
initialization: all_four_v1_final_latents_not_optimizer_resume
|
| 147 |
+
fp32_stage:
|
| 148 |
+
steps: 800
|
| 149 |
+
maximum_learning_rate: 0.03
|
| 150 |
+
minimum_learning_rate: 0.003
|
| 151 |
+
bf16_stage:
|
| 152 |
+
steps: 1200
|
| 153 |
+
maximum_learning_rate: 0.015
|
| 154 |
+
minimum_learning_rate: 0.0015
|
| 155 |
+
thresholds:
|
| 156 |
+
minimum_objective_improvement_fraction: 0.20
|
| 157 |
+
minimum_median_objective_improvement_fraction: 0.40
|
| 158 |
+
maximum_waveform_mae: 0.06
|
| 159 |
+
minimum_correlation: 0.50
|
| 160 |
+
minimum_si_sdr_db: -1.0
|
| 161 |
+
maximum_latent_rmse: null
|
| 162 |
+
high_fidelity_minimum_si_sdr_db: 20.0
|
| 163 |
+
high_fidelity_minimum_unscaled_snr_db: 18.0
|
| 164 |
+
high_fidelity_maximum_loudness_error_db: 0.5
|
| 165 |
+
high_fidelity_maximum_stereo_correlation_error: 0.05
|
| 166 |
+
publication:
|
| 167 |
+
root_mode: 493
|
| 168 |
+
experiment_directory_mode: 448
|
| 169 |
+
file_mode: 420
|
| 170 |
+
manifest_written_last: true
|
| 171 |
+
atomic_directory_rename: true
|
| 172 |
+
notes:
|
| 173 |
+
capability: continuous Flow-VAE latent-only continuation; not WAV-to-RVQ encoding
|
| 174 |
+
optimizer_resume: false
|
| 175 |
+
quality_claim_policy: preserve v1.1 feasibility/high-fidelity gates; engineering progress is separately labeled
|
configs/learned-audio-continuation-v1.yaml
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.learned-audio-continuation-config.v1
|
| 2 |
+
capability: "continuous-latent local continuation; non-native-token; not long-song"
|
| 3 |
+
sample_rate: 44100
|
| 4 |
+
samples_per_window: 44032
|
| 5 |
+
target_parameterization: target_minus_repeat_tail
|
| 6 |
+
generated_latent_parameterization: repeat_tail_plus_generated_residual
|
| 7 |
+
corpus:
|
| 8 |
+
manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 9 |
+
splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
| 10 |
+
semantic_digest: 6cd403e9cd25e61d062af6caa9c6b136b751c7f668f1af7d0972989963ac3452
|
| 11 |
+
train_count: 542
|
| 12 |
+
validation_count: 68
|
| 13 |
+
heldout_count: 68
|
| 14 |
+
guard_samples: 220500
|
| 15 |
+
crop_namespace: music3lab.learned-continuation.crop.v1
|
| 16 |
+
source_exclusive: true
|
| 17 |
+
one_pair_per_source: true
|
| 18 |
+
projector:
|
| 19 |
+
checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 20 |
+
source_gate: REJECTED_teacher_latent_nmse_regression
|
| 21 |
+
champion_eligible: false
|
| 22 |
+
frozen: true
|
| 23 |
+
latent_shape: [128, 86]
|
| 24 |
+
condition_shape: [86, 2048]
|
| 25 |
+
condition_repeat: 16
|
| 26 |
+
terminal_zero_trim: exact_only
|
| 27 |
+
target_visible_to_conditioner: false
|
| 28 |
+
adapter:
|
| 29 |
+
flow_layers: 36
|
| 30 |
+
hidden_size: 2048
|
| 31 |
+
rank: 4
|
| 32 |
+
scale: 1.0
|
| 33 |
+
expected_trainable_parameters: 1769472
|
| 34 |
+
base_flow_frozen: true
|
| 35 |
+
vocoder_frozen: true
|
| 36 |
+
training:
|
| 37 |
+
steps: 1200
|
| 38 |
+
batch_size: 16
|
| 39 |
+
seed: 260815
|
| 40 |
+
maximum_learning_rate: 0.00005
|
| 41 |
+
minimum_learning_rate: 0.000005
|
| 42 |
+
beta1: 0.9
|
| 43 |
+
beta2: 0.999
|
| 44 |
+
epsilon: 0.00000001
|
| 45 |
+
weight_decay: 0.0
|
| 46 |
+
gradient_clip_norm: 1.0
|
| 47 |
+
validation_interval_steps: 100
|
| 48 |
+
gradient_checkpointing: true
|
| 49 |
+
compute_dtype: bfloat16
|
| 50 |
+
parameter_dtype: float32
|
| 51 |
+
inference:
|
| 52 |
+
euler_steps: 30
|
| 53 |
+
guidance_scale: 1.7
|
| 54 |
+
noise_seed: 73021
|
| 55 |
+
overlap_samples: 1024
|
| 56 |
+
evaluation:
|
| 57 |
+
same_noise_across_conditions: true
|
| 58 |
+
condition_baselines: [zero_context, unrelated_context]
|
| 59 |
+
direct_baselines: [repeat_tail, roll_tail]
|
| 60 |
+
strict_median_latent_nmse_beats_every_baseline: true
|
| 61 |
+
strict_median_audio_ruler_beats_every_baseline: true
|
| 62 |
+
benchmark_batch_sizes: [4, 8, 16]
|
| 63 |
+
seam:
|
| 64 |
+
metric_domain: composed_output
|
| 65 |
+
derivative_absolute_floor: 0.00001
|
| 66 |
+
rms_absolute_floor: 0.0001
|
| 67 |
+
maximum_median_boundary_derivative_ratio: 2.0
|
| 68 |
+
maximum_median_overlap_rms_log_error: 1.5
|
| 69 |
+
claims:
|
| 70 |
+
native_tokens: false
|
| 71 |
+
text_conditioning: false
|
| 72 |
+
captured_condition: false
|
| 73 |
+
future_audio_conditioning: false
|
| 74 |
+
arbitrary_wav_local_continuation: true
|
| 75 |
+
long_song_generalization: false
|
| 76 |
+
specialist_promoted: false
|
| 77 |
+
handcrafted_baseline_reclassified: false
|
configs/long-reference-style-v1.yaml
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.long-reference-style-config.v1
|
| 2 |
+
capability: long_reference_internal_text_bridge_plus_postrender_ranking
|
| 3 |
+
model_id: MiniMaxAI/MiniMax-Music3
|
| 4 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
channels: 2
|
| 8 |
+
frame_rate: 25
|
| 9 |
+
chunk_frames: 200
|
| 10 |
+
chunk_hop: 100
|
| 11 |
+
inference_steps: 12
|
| 12 |
+
guidance_scale: 1.7
|
| 13 |
+
profile_samples: 44032
|
| 14 |
+
profile_maximum_crops: 8
|
| 15 |
+
specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 16 |
+
specialist_role: REJECTED_EXTERNAL_SPECIALIST_STYLE_SCORER_ONLY
|
| 17 |
+
policies:
|
| 18 |
+
fast: {candidates: 2, duration_seconds: 30.0}
|
| 19 |
+
balanced: {candidates: 4, duration_seconds: 60.0}
|
| 20 |
+
maximum: {candidates: 8, duration_seconds: 90.0}
|
| 21 |
+
independent_ar_flow_seed_supported: false
|
| 22 |
+
direct_model_audio_conditioning: false
|
| 23 |
+
native_negative_prompt_used: false
|
| 24 |
+
full_semantic_style_transfer: false
|
| 25 |
+
artistic_quality_claim: false
|
| 26 |
+
vocal_absence_guaranteed: false
|
configs/native-state-distill.yaml
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.native-state-distill-config.v1
|
| 2 |
+
capability: captured_music3_multiclip_only
|
| 3 |
+
source:
|
| 4 |
+
teacher_manifest_sha256: 6ceffbe5aab6ea808f66ada57bbe3e91244f96db15f5038040269afbc47b3071
|
| 5 |
+
primary_splits_only: true
|
| 6 |
+
diagnostic_clips_excluded: true
|
| 7 |
+
teacher:
|
| 8 |
+
semantics: official_post_cfg_conditional_top50_semantic_mask_sampling_distribution
|
| 9 |
+
storage: finite_normalized_float32_log_probabilities
|
| 10 |
+
temperature: 1.0
|
| 11 |
+
audio_code_offset: 151675
|
| 12 |
+
semantic_vocab_size: 16384
|
| 13 |
+
cfg_scale: 1.5
|
| 14 |
+
conditional_top_k: 50
|
| 15 |
+
sampling_top_k: 50
|
| 16 |
+
initialization:
|
| 17 |
+
description: selected multiclip checkpoint before distillation
|
| 18 |
+
checkpoint_sha256: ed8505a00438a3d3b1a605584313f6214dee37e13dba6f07aab4a5c42d700c43
|
| 19 |
+
training:
|
| 20 |
+
seed: 929292
|
| 21 |
+
steps: 5000
|
| 22 |
+
batch_size: 8
|
| 23 |
+
validation_interval_steps: 250
|
| 24 |
+
maximum_learning_rate: 0.0003
|
| 25 |
+
minimum_learning_rate: 0.00003
|
| 26 |
+
gradient_clip_norm: 10.0
|
| 27 |
+
weight_decay: 0.0
|
| 28 |
+
optimization_scope: semantic_head_only_to_preserve_native_state_heads
|
| 29 |
+
loss_weights:
|
| 30 |
+
c0_kl: 5.0
|
| 31 |
+
feedback: 2.0
|
| 32 |
+
global_hidden: 0.5
|
| 33 |
+
residual: 0.0
|
| 34 |
+
selection:
|
| 35 |
+
dataset: primary_unseen_prompt_validation_only
|
| 36 |
+
objective: minimum_c0_distillation_kl
|
| 37 |
+
state_nmse_regression_tolerance: 0.000001
|
| 38 |
+
terminal_gate:
|
| 39 |
+
heldout_hard_c0_ce_maximum: 9.604060527839234
|
| 40 |
+
heldout_kl_improvement_fraction_minimum: 0.10
|
| 41 |
+
heldout_feedback_nmse_maximum: 1.114844
|
| 42 |
+
heldout_global_nmse_maximum: 0.844784
|
| 43 |
+
rounding_tolerance: 0.000001
|
| 44 |
+
native_tokenizer: false
|
| 45 |
+
arbitrary_external_music: false
|
| 46 |
+
generalization_claim: false
|
configs/native-state-multiclip.yaml
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.native-state-multiclip-config.v1
|
| 2 |
+
capability: captured_music3_multiclip_only
|
| 3 |
+
model:
|
| 4 |
+
width: 96
|
| 5 |
+
heads: 4
|
| 6 |
+
conformer_layers: 8
|
| 7 |
+
mel_bins: 80
|
| 8 |
+
conv_kernel: 15
|
| 9 |
+
residual_heads: disabled
|
| 10 |
+
capture:
|
| 11 |
+
duration_seconds: 1.0
|
| 12 |
+
inference_steps: 12
|
| 13 |
+
lyrics: "[instrumental]"
|
| 14 |
+
fresh_prompts:
|
| 15 |
+
train_a: "Instrumental UK garage, 132 BPM, swung drums, deep sub bass, bright chopped chords."
|
| 16 |
+
train_b: "Instrumental ambient electronic, 90 BPM, evolving pads, soft pulse, spacious granular texture."
|
| 17 |
+
validation_c: "Instrumental Latin electronic, 110 BPM, syncopated hand percussion, warm bass, vivid mallet motif."
|
| 18 |
+
heldout_d: "Instrumental acoustic folk, 76 BPM, fingerpicked guitar, brushed percussion, intimate strings."
|
| 19 |
+
training:
|
| 20 |
+
seed: 919191
|
| 21 |
+
steps: 10000
|
| 22 |
+
batch_size: 8
|
| 23 |
+
validation_interval_steps: 500
|
| 24 |
+
maximum_learning_rate: 0.0003
|
| 25 |
+
minimum_learning_rate: 0.00003
|
| 26 |
+
gradient_clip_norm: 10.0
|
| 27 |
+
weight_decay: 0.0
|
| 28 |
+
loss_weights:
|
| 29 |
+
c0: 5.0
|
| 30 |
+
feedback: 2.0
|
| 31 |
+
global_hidden: 0.5
|
| 32 |
+
residual: 0.0
|
| 33 |
+
selection:
|
| 34 |
+
dataset: primary_unseen_prompt_validation_only
|
| 35 |
+
objective: stage0_weighted_total
|
| 36 |
+
scale_up_gate:
|
| 37 |
+
heldout_feedback_nmse_improvement_minimum: 0.05
|
| 38 |
+
heldout_global_nmse_improvement_minimum: 0.05
|
| 39 |
+
heldout_c0_ce_maximum_margin_below_uniform: 0.1
|
| 40 |
+
train_vs_heldout_error_ratio_maximum: 2.0
|
| 41 |
+
error_ratio_definition: heldout_error_divided_by_train_error
|
| 42 |
+
native_tokenizer: false
|
| 43 |
+
arbitrary_external_music: false
|
| 44 |
+
generalization_claim: false
|
configs/native-state-stage0.yaml
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.native-state-stage0.v1
|
| 2 |
+
capability: one_clip_music3_native_state_posterior_alignment_proof
|
| 3 |
+
teacher:
|
| 4 |
+
source_case: existing_native_token_adapter_train_seed_1000
|
| 5 |
+
prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
|
| 6 |
+
lyrics: "[instrumental]"
|
| 7 |
+
seed: 1000
|
| 8 |
+
duration_seconds: 1.0
|
| 9 |
+
inference_steps: 12
|
| 10 |
+
first_sampled_row_is_priming: true
|
| 11 |
+
emitted_frame_count: 25
|
| 12 |
+
model:
|
| 13 |
+
width: 96
|
| 14 |
+
heads: 4
|
| 15 |
+
conformer_layers: 8
|
| 16 |
+
mel_bins: 80
|
| 17 |
+
conv_kernel: 15
|
| 18 |
+
residual_heads: disabled
|
| 19 |
+
loss:
|
| 20 |
+
c0: 5.0
|
| 21 |
+
feedback: 2.0
|
| 22 |
+
global_hidden: 0.5
|
| 23 |
+
residual: 0.0
|
| 24 |
+
alignment:
|
| 25 |
+
minimum_offset: -8
|
| 26 |
+
maximum_offset: 8
|
| 27 |
+
calibration_steps: 80
|
| 28 |
+
calibration_repeats: 2
|
| 29 |
+
full_overfit_steps: 6000
|
| 30 |
+
learning_rate: 0.001
|
| 31 |
+
acceptance:
|
| 32 |
+
c0_exact_frames: 25
|
| 33 |
+
feedback_normalized_mse_maximum: 0.001
|
| 34 |
+
feedback_cosine_minimum: 0.999
|
| 35 |
+
global_hidden_normalized_mse_maximum: 0.001
|
| 36 |
+
global_hidden_cosine_minimum: 0.999
|
| 37 |
+
native_tokenizer: false
|
| 38 |
+
arbitrary_external_music: false
|
| 39 |
+
generalization_claim: false
|
configs/native-state-stage1-local.yaml
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.native-stage1-local-config.v1
|
| 2 |
+
capability: captured_music3_only
|
| 3 |
+
corpus:
|
| 4 |
+
train: 800
|
| 5 |
+
validation: 112
|
| 6 |
+
heldout: 112
|
| 7 |
+
frames: 25
|
| 8 |
+
lyrics: "[instrumental]"
|
| 9 |
+
inference_steps: 12
|
| 10 |
+
augmentation:
|
| 11 |
+
checkpoint_every_clips: 256
|
| 12 |
+
generation_calls: 0
|
| 13 |
+
training:
|
| 14 |
+
seed: 969696
|
| 15 |
+
steps: 5000
|
| 16 |
+
batch_size: 8
|
| 17 |
+
validation_interval_steps: 250
|
| 18 |
+
maximum_learning_rate: 0.0003
|
| 19 |
+
minimum_learning_rate: 0.00003
|
| 20 |
+
gradient_clip_norm: 10.0
|
| 21 |
+
weight_decay: 0.0
|
| 22 |
+
hidden_nmse_weight: 1.0
|
| 23 |
+
hidden_cosine_weight: 1.0
|
| 24 |
+
guided_kl_weight: 0.25
|
| 25 |
+
hard_ce_weight: 1.0
|
| 26 |
+
selection: lowest_validation_mean_residual_ce_subject_to_finite_and_stage0_identity
|
| 27 |
+
bootstrap:
|
| 28 |
+
seed: 979797
|
| 29 |
+
resamples: 2000
|
| 30 |
+
confidence: 0.95
|
| 31 |
+
native_tokenizer: false
|
| 32 |
+
arbitrary_external_music: false
|
configs/native-state-stage1.yaml
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.native-stage1-config.v1
|
| 2 |
+
capability: captured_music3_only
|
| 3 |
+
corpus:
|
| 4 |
+
train: 800
|
| 5 |
+
validation: 112
|
| 6 |
+
heldout: 112
|
| 7 |
+
train_prompt_ids: 25
|
| 8 |
+
validation_prompt_ids: 4
|
| 9 |
+
heldout_prompt_ids: 4
|
| 10 |
+
duration_seconds: 1.0
|
| 11 |
+
inference_steps: 12
|
| 12 |
+
lyrics: "[instrumental]"
|
| 13 |
+
shard_size: 32
|
| 14 |
+
training:
|
| 15 |
+
seed: 939393
|
| 16 |
+
steps: 5000
|
| 17 |
+
batch_size: 8
|
| 18 |
+
validation_interval_steps: 250
|
| 19 |
+
maximum_learning_rate: 0.0003
|
| 20 |
+
minimum_learning_rate: 0.00003
|
| 21 |
+
gradient_clip_norm: 10.0
|
| 22 |
+
weight_decay: 0.0
|
| 23 |
+
scheduled_sampling_teacher_fraction: 0.4
|
| 24 |
+
hard_ce_weight: 1.0
|
| 25 |
+
soft_kl_weight: 0.25
|
| 26 |
+
selection: lowest_validation_mean_residual_hard_ce_subject_to_finite_and_exact_stage0
|
| 27 |
+
bootstrap:
|
| 28 |
+
seed: 949494
|
| 29 |
+
resamples: 2000
|
| 30 |
+
confidence: 0.95
|
| 31 |
+
terminal:
|
| 32 |
+
each_depth_ce_improvement: 0.03
|
| 33 |
+
mean_ce_improvement: 0.05
|
| 34 |
+
top1_minimum_depths: 5
|
| 35 |
+
top1_improvement_pp: 1.0
|
| 36 |
+
autoregressive_vs_h_only_improvement: 0.02
|
| 37 |
+
full_row_paired_bootstrap_lower_ci_strictly_above: 0.0
|
| 38 |
+
native_tokenizer: false
|
| 39 |
+
arbitrary_external_music: false
|
| 40 |
+
generalization_claim: false
|
configs/native-token-adapter-v1.yaml
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.native-token-adapter-config.v1
|
| 2 |
+
capability: captured_music3_waveform_to_rvq_row_prediction_only
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
flow_encoder_checkpoint_sha256: e8d32b6fb60b8f04f0ad1641bcf45a03222d2dce90cc7fd1dc4309228b284027
|
| 7 |
+
teacher:
|
| 8 |
+
prompt: "Instrumental French house, 126 BPM, E minor, filtered disco loop, punchy kick and warm bass."
|
| 9 |
+
lyrics: "[instrumental]"
|
| 10 |
+
duration_seconds: 1.0
|
| 11 |
+
inference_steps: 12
|
| 12 |
+
guidance_scale: 1.7
|
| 13 |
+
train_seed_start: 1000
|
| 14 |
+
train_count: 64
|
| 15 |
+
validation_seed_start: 2000
|
| 16 |
+
validation_count: 16
|
| 17 |
+
heldout_seed_start: 3000
|
| 18 |
+
heldout_count: 16
|
| 19 |
+
training:
|
| 20 |
+
steps: 2000
|
| 21 |
+
batch_size: 8
|
| 22 |
+
seed: 515151
|
| 23 |
+
maximum_learning_rate: 0.0003
|
| 24 |
+
minimum_learning_rate: 0.00003
|
| 25 |
+
weight_decay: 0.0001
|
| 26 |
+
gradient_clip_norm: 1.0
|
| 27 |
+
validation_interval_steps: 100
|
| 28 |
+
evaluation:
|
| 29 |
+
predicted_render_seed: 9091
|
| 30 |
+
top_k: 5
|
| 31 |
+
native_tokenizer: false
|
| 32 |
+
arbitrary_external_music: false
|
| 33 |
+
generalization_claim: false
|
configs/objective-only-v1.yaml
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.objective-only-config.v1
|
| 2 |
+
corpus:
|
| 3 |
+
corpus_label: opaque-user-audio-objective-v1
|
| 4 |
+
source_count: 8
|
| 5 |
+
raw_file_mode: 292
|
| 6 |
+
crop_file_mode: 292
|
| 7 |
+
directory_mode: 493
|
| 8 |
+
manifest_file_mode: 420
|
| 9 |
+
preserve_human_metadata: false
|
| 10 |
+
source_identity: raw_sha256
|
| 11 |
+
split:
|
| 12 |
+
algorithm: prior_exposure_then_lexicographic_sha256_v1
|
| 13 |
+
prior_exposure_source_ids:
|
| 14 |
+
- f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
|
| 15 |
+
calibration_count: 1
|
| 16 |
+
validation_count: 3
|
| 17 |
+
heldout_count: 4
|
| 18 |
+
heldout_sealed: true
|
| 19 |
+
additions_require_new_schema: true
|
| 20 |
+
crop:
|
| 21 |
+
algorithm: sha256_domain_uint64_mod_interior_v1
|
| 22 |
+
domain: "music3lab.objective-only.crop.v1"
|
| 23 |
+
output_sample_rate: 44100
|
| 24 |
+
output_channels: 2
|
| 25 |
+
output_frames: 44032
|
| 26 |
+
guard_frames: 220500
|
| 27 |
+
operation_order: resample_full_source_then_trim_output_frames
|
| 28 |
+
canonicalizer:
|
| 29 |
+
executable: /usr/bin/ffmpeg
|
| 30 |
+
executable_sha256: 36d94a605d612e4090d1b8aec889d0c0801c6eafb1593c90f5c0dfd2e2966a45
|
| 31 |
+
version_line: ffmpeg version 4.4.2-0ubuntu0.22.04.1 Copyright (c) 2000-2021 the FFmpeg developers
|
| 32 |
+
input_transport: retained_source_bytes_via_stdin
|
| 33 |
+
output_encoding: stereo_interleaved_little_endian_float32
|
| 34 |
+
sources:
|
| 35 |
+
- source_id: 054f93d3933c4f3e4839af1d901b71f6ba0d80b5dc6abd80776a22545fca13a9
|
| 36 |
+
raw_size: 2208136
|
| 37 |
+
container: mp3
|
| 38 |
+
codec: mp3
|
| 39 |
+
sample_rate: 44100
|
| 40 |
+
channels: 2
|
| 41 |
+
time_base: 1/14112000
|
| 42 |
+
duration_ticks: 1298349864
|
| 43 |
+
canonical_frames: 4057344
|
| 44 |
+
canonical_full_sha256: 8587aabd20d0d7709b31b1e8178519fe9f4596e37bb9e2e3a2e73e1b90cfba80
|
| 45 |
+
split: validation
|
| 46 |
+
crop_start_frame: 784654
|
| 47 |
+
crop_interleaved_sha256: afb4e733fa35ce4651776f4fa8fcb3bf37c80e65d5ddbd3a90c804a860d6e5a9
|
| 48 |
+
- source_id: 5c6ea28a706eacea653309ab40548ca324f7dd1f5cc1707a16d4ef93e46ec310
|
| 49 |
+
raw_size: 2344078
|
| 50 |
+
container: m4a
|
| 51 |
+
codec: aac
|
| 52 |
+
sample_rate: 48000
|
| 53 |
+
channels: 2
|
| 54 |
+
time_base: 1/48000
|
| 55 |
+
duration_ticks: 6847488
|
| 56 |
+
canonical_frames: 6289190
|
| 57 |
+
canonical_full_sha256: 939a017c0c4554f5b072f6dd6ff00a73c0142478168db2c4f7bc7ceef0d60eaf
|
| 58 |
+
split: validation
|
| 59 |
+
crop_start_frame: 4591657
|
| 60 |
+
crop_interleaved_sha256: 457d2ce0dfdb35963d8835d194943c28f4c99146637a8299a98180a90cdb7888
|
| 61 |
+
- source_id: 994f9a6b74ebc3ed9bd3d25433b77c0415c9f53a8c73441f2891b46211321e14
|
| 62 |
+
raw_size: 33375668
|
| 63 |
+
container: wav
|
| 64 |
+
codec: pcm_s16le
|
| 65 |
+
sample_rate: 48000
|
| 66 |
+
channels: 2
|
| 67 |
+
time_base: 1/48000
|
| 68 |
+
duration_ticks: 8343871
|
| 69 |
+
canonical_frames: 7665932
|
| 70 |
+
canonical_full_sha256: 49cb8dcaa9183bbca46db61a63d5ef61d56e2e5bf170853c00a407074138f2f0
|
| 71 |
+
split: validation
|
| 72 |
+
crop_start_frame: 2163717
|
| 73 |
+
crop_interleaved_sha256: 8f0ea8da159164cc03e231e39e15ce7259e62ad03236cfb96a6ead91a05db6e7
|
| 74 |
+
- source_id: 9b3291618c7f00a988fa1093243c0c575fd593d9c1d952db3b44a2326424116d
|
| 75 |
+
raw_size: 33180184
|
| 76 |
+
container: wav
|
| 77 |
+
codec: pcm_s16le
|
| 78 |
+
sample_rate: 48000
|
| 79 |
+
channels: 2
|
| 80 |
+
time_base: 1/48000
|
| 81 |
+
duration_ticks: 8295000
|
| 82 |
+
canonical_frames: 7621032
|
| 83 |
+
canonical_full_sha256: 9e7b8a4eadd86be966ce3bc9d799eddbffda4e6720ac6f6bffb06527b5645f45
|
| 84 |
+
split: heldout_test
|
| 85 |
+
crop_start_frame: 6372962
|
| 86 |
+
crop_interleaved_sha256: 07cbed859a775f9e9296066386dedf068916022e488140891ab2d4ea561d451f
|
| 87 |
+
- source_id: c2368b62f03fb46455d429524e3613d2ba8746d1e4c3435fef73e62369b5be9d
|
| 88 |
+
raw_size: 2284542
|
| 89 |
+
container: m4a
|
| 90 |
+
codec: aac
|
| 91 |
+
sample_rate: 48000
|
| 92 |
+
channels: 2
|
| 93 |
+
time_base: 1/48000
|
| 94 |
+
duration_ticks: 6648832
|
| 95 |
+
canonical_frames: 6106674
|
| 96 |
+
canonical_full_sha256: 9cf7287782a1548a6fbc98cd0aebb689c1cd6675c404455bc75052ce8787e23b
|
| 97 |
+
split: heldout_test
|
| 98 |
+
crop_start_frame: 232371
|
| 99 |
+
crop_interleaved_sha256: c63f87185e7e1337efef296aa5da0a725ee694264647bcde6292673fd7abb692
|
| 100 |
+
- source_id: e182bd207c5a02276cc04f0b932ac47842ab61855b33d762bcfa344042b75c91
|
| 101 |
+
raw_size: 31204096
|
| 102 |
+
container: wav
|
| 103 |
+
codec: pcm_s16le
|
| 104 |
+
sample_rate: 48000
|
| 105 |
+
channels: 2
|
| 106 |
+
time_base: 1/48000
|
| 107 |
+
duration_ticks: 7800000
|
| 108 |
+
canonical_frames: 7166250
|
| 109 |
+
canonical_full_sha256: eeece5e958b727c483396d93df29a51909f6edfe234885410888f82f5b9530e6
|
| 110 |
+
split: heldout_test
|
| 111 |
+
crop_start_frame: 4023033
|
| 112 |
+
crop_interleaved_sha256: 84804551646a84f958d30e027575fcdd9001757bd06dae13b2f7ede5480c79f7
|
| 113 |
+
- source_id: f4e6269f3302d80c956a5b09c2c636324091b6642e370811e7fab126d521a6e5
|
| 114 |
+
raw_size: 31437094
|
| 115 |
+
container: wav
|
| 116 |
+
codec: pcm_s32le
|
| 117 |
+
sample_rate: 48000
|
| 118 |
+
channels: 2
|
| 119 |
+
time_base: 1/48000
|
| 120 |
+
duration_ticks: 3929624
|
| 121 |
+
canonical_frames: 3610343
|
| 122 |
+
canonical_full_sha256: 392112ff198a7f2fa5ffb92f6083a21e2c7532b826fb5afcb86acf6d72f38d8c
|
| 123 |
+
split: calibration
|
| 124 |
+
crop_start_frame: 1103007
|
| 125 |
+
crop_interleaved_sha256: d2afcf0f0f034c32ad0d044cfd7cb89dde634aa97d2c7efa1f81f82b9969cc52
|
| 126 |
+
- source_id: ff777c257a9f8044bbe08eb2933d6f4073a22770cf04ae527b8fe768982e6fea
|
| 127 |
+
raw_size: 13153696
|
| 128 |
+
container: wav
|
| 129 |
+
codec: pcm_s16le
|
| 130 |
+
sample_rate: 44100
|
| 131 |
+
channels: 2
|
| 132 |
+
time_base: 1/44100
|
| 133 |
+
duration_ticks: 3288354
|
| 134 |
+
canonical_frames: 3288354
|
| 135 |
+
canonical_full_sha256: da3cd2960f30d0c58845ce3b6ca1fe807b5c9a237d34a9edf13ccd96b0fea3a6
|
| 136 |
+
split: heldout_test
|
| 137 |
+
crop_start_frame: 1360904
|
| 138 |
+
crop_interleaved_sha256: d85d4d805e99e305b571ad2a464e7bd1a80cfffb02249129e5441aebdeda7a23
|
| 139 |
+
metric_policy:
|
| 140 |
+
mandatory_boards: [signal, reference, boundary, diversity]
|
| 141 |
+
signal:
|
| 142 |
+
minimum_rms: 0.0001
|
| 143 |
+
maximum_peak: 1.5
|
| 144 |
+
maximum_clipped_fraction: 0.05
|
| 145 |
+
reference:
|
| 146 |
+
maximum_median_waveform_mae: 0.06
|
| 147 |
+
minimum_median_correlation: 0.5
|
| 148 |
+
minimum_median_si_sdr_db: -1.0
|
| 149 |
+
minimum_median_unscaled_snr_db: -1.0
|
| 150 |
+
boundary:
|
| 151 |
+
maximum_edge_derivative: 0.5
|
| 152 |
+
maximum_edge_rms_ratio: 4.0
|
| 153 |
+
diversity:
|
| 154 |
+
minimum_unique_fraction: 1.0
|
| 155 |
+
minimum_pairwise_feature_distance: 0.0001
|
| 156 |
+
optional_evaluators:
|
| 157 |
+
clap_alignment: false
|
| 158 |
+
frechet_audio_distance: false
|
| 159 |
+
learned_music_quality: false
|
| 160 |
+
candidate_policy:
|
| 161 |
+
selection_allowed_splits: [calibration, validation]
|
| 162 |
+
heldout_split: heldout_test
|
| 163 |
+
selection_must_be_locked: true
|
| 164 |
+
missing_mandatory_result: INSUFFICIENT_EVIDENCE
|
| 165 |
+
publication:
|
| 166 |
+
root_mode: 493
|
| 167 |
+
leaf_mode: 448
|
| 168 |
+
file_mode: 420
|
| 169 |
+
manifest_written_last: true
|
| 170 |
+
atomic_directory_rename: true
|
configs/prefix-seed-search-v1.yaml
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.prefix-seed-search-config.v1
|
| 2 |
+
capability: least_measurable_splice_discontinuity_among_four_captured_prefix_candidates
|
| 3 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 4 |
+
expected_base_id: 05c0cdd43a6374b6959cd561e8de4cd6a3327133f8763312d41b1d2b6549c53d
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
source:
|
| 7 |
+
prepend_project_commit: 82795b470538ec1c08793a559016f1f9fcd38ab8
|
| 8 |
+
prepend_config_file_sha256: 995f216302b682e495aaf3a5a7499ca1d595f1a31e7c7c43ddee7558fb1d6c48
|
| 9 |
+
prepend_config_semantic_digest: ee16251758454669bc58710752ce976f15739d4f242fb7a3f5d7e206ab46b554
|
| 10 |
+
prepend_bundle_semantic_digest: 181832f93cb0a41a12c07fdf54ddf817ac4f20fa1586e01f7956b8e0da45bdec
|
| 11 |
+
prepend_metrics_semantic_digest: 121386375685bf599c406ed434808b42af4bf6f3101d7bfcc19554bba3395f7b
|
| 12 |
+
prepend_metrics_file_sha256: 6d9926f9c44634f9fabb4afe51e274d39cb0d1a6e2f4f7ae0285d935eb5ad0f3
|
| 13 |
+
prepend_adapter_file_sha256: 0048d3d727f0faeea53fcc136f777917e58435de8cba8a35b293cae6ad5aab28
|
| 14 |
+
prepend_adapter_state_sha256: a60c74c557035a19ebb4138b683015e4f4358b9881f28cdb775a16c2fc52ba1a
|
| 15 |
+
teacher_v2_semantic_digest: 0b8fb727a380b5a57373f12fb4ffd31eca65c2bdf855d4eee404c2380b3e5607
|
| 16 |
+
heldout_index: 0
|
| 17 |
+
heldout_seed: 3000
|
| 18 |
+
search:
|
| 19 |
+
seeds: [101, 103, 107, 109]
|
| 20 |
+
sequential_single_model_load: true
|
| 21 |
+
prefix_mask_seed: 880301
|
| 22 |
+
expected_prefix_frames: 22
|
| 23 |
+
expected_latent_frames: 86
|
| 24 |
+
frame_samples: 512
|
| 25 |
+
expected_audio_samples: 44032
|
| 26 |
+
channels: 2
|
| 27 |
+
sampling_rate: 44100
|
| 28 |
+
euler_steps: 30
|
| 29 |
+
guidance_scale: 1.7
|
| 30 |
+
gates:
|
| 31 |
+
minimum_rms: 0.0001
|
| 32 |
+
maximum_abs_peak: 1.5
|
| 33 |
+
clipped_abs_threshold: 1.0
|
| 34 |
+
maximum_clipped_fraction: 0.05
|
| 35 |
+
require_finite: true
|
| 36 |
+
require_expected_length: true
|
| 37 |
+
require_known_suffix_exact: true
|
| 38 |
+
require_unique_audio_sha256: true
|
| 39 |
+
selection:
|
| 40 |
+
join_derivative_half_window_samples: 512
|
| 41 |
+
edge_rms_window_samples: 1024
|
| 42 |
+
edge_rms_epsilon: 0.00000001
|
| 43 |
+
lexicographic_fields:
|
| 44 |
+
- join_max_abs_derivative
|
| 45 |
+
- join_edge_rms_ratio
|
| 46 |
+
- seed
|
| 47 |
+
publication:
|
| 48 |
+
evidence_root_mode: 493
|
| 49 |
+
evidence_file_mode: 420
|
| 50 |
+
manifest_written_last: true
|
| 51 |
+
atomic_directory_rename: true
|
| 52 |
+
artistic_quality_claim: false
|
| 53 |
+
stock_win_claim: false
|
| 54 |
+
|
| 55 |
+
# The selected result is only the least measurable splice discontinuity among
|
| 56 |
+
# these four fixed candidates. It is not an artistic-quality or stock-win claim.
|
configs/prompt-free-v1.yaml
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.prompt-free-config.v1
|
| 2 |
+
capability: prompt_free_internal_text_compatibility_bridge
|
| 3 |
+
model_id: MiniMaxAI/MiniMax-Music3
|
| 4 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
channels: 2
|
| 8 |
+
inference_steps: 12
|
| 9 |
+
guidance_scale: 1.7
|
| 10 |
+
minimum_duration_seconds: 1.0
|
| 11 |
+
maximum_duration_seconds: 30.0
|
| 12 |
+
lyrics_literal: "[instrumental]"
|
| 13 |
+
source_prior_count: 32
|
| 14 |
+
text_origin: internal_generated
|
| 15 |
+
null_conditioning: false
|
| 16 |
+
artistic_quality_claim: false
|
| 17 |
+
rendered_diversity_claim: false
|
| 18 |
+
guaranteed_no_vocals: false
|
| 19 |
+
arbitrary_audio_reference_conditioning: false
|
configs/promptfree-bestofn-v1.yaml
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.promptfree-bestofn-config.v1
|
| 2 |
+
capability: prompt_free_long_form_best_of_n_internal_text_bridge
|
| 3 |
+
model_id: MiniMaxAI/MiniMax-Music3
|
| 4 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
channels: 2
|
| 8 |
+
frame_rate: 25
|
| 9 |
+
chunk_frames: 200
|
| 10 |
+
chunk_hop: 100
|
| 11 |
+
inference_steps: 12
|
| 12 |
+
guidance_scale: 1.7
|
| 13 |
+
source_prior_count: 32
|
| 14 |
+
duration_minimum_seconds: 1
|
| 15 |
+
duration_maximum_seconds: 360
|
| 16 |
+
diversity_epsilon: 1.0e-6
|
| 17 |
+
policies:
|
| 18 |
+
fast:
|
| 19 |
+
candidates: 2
|
| 20 |
+
default_duration_seconds: 30.0
|
| 21 |
+
balanced:
|
| 22 |
+
candidates: 4
|
| 23 |
+
default_duration_seconds: 60.0
|
| 24 |
+
maximum:
|
| 25 |
+
candidates: 8
|
| 26 |
+
default_duration_seconds: 90.0
|
| 27 |
+
internal_plan_only: true
|
| 28 |
+
independent_ar_flow_seed_supported: false
|
| 29 |
+
artistic_quality_claim: false
|
configs/reference-guided-append-v1.yaml
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.reference-guided-append-config.v1
|
| 2 |
+
capability: reference_guided_append
|
| 3 |
+
sample_rate: 44100
|
| 4 |
+
channels: 2
|
| 5 |
+
candidate_frames: 352768
|
| 6 |
+
crossfade_frames: 2048
|
| 7 |
+
seam_weight: 0.15
|
| 8 |
+
silence_rms: 0.0001
|
| 9 |
+
clipped_fraction_limit: 0.05
|
| 10 |
+
source_copy_correlation: 0.75
|
| 11 |
+
model_generation: false
|
| 12 |
+
native_history_conditioning: false
|
| 13 |
+
semantic_style_transfer: false
|
| 14 |
+
native_negative_prompt: false
|
configs/reference-ranked-pool-v1.yaml
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.reference-ranked-pool-config.v1
|
| 2 |
+
capability: reference_ranked_long_form_pool_not_direct_audio_conditioning
|
| 3 |
+
source_audio: /home/ubuntu/minimax-user-audio/incoming/loveonme_x_osh.wav
|
| 4 |
+
source_audio_sha256: ff777c257a9f8044bbe08eb2933d6f4073a22770cf04ae527b8fe768982e6fea
|
| 5 |
+
pool_root: /home/ubuntu/minimax-promptfree-bestofn-evidence/maximum90-c19c0c5
|
| 6 |
+
retained_candidate_ids: ["002", "003", "007"]
|
| 7 |
+
retained_candidate_wav_sha256:
|
| 8 |
+
"002": 72855ef9cf1e77e5aa718e956e27cec7a23b8e55b9ecb1b53229011a48409ddd
|
| 9 |
+
"003": 29e2c8afc5dd00c39bd8a79c278829d4ea8450703e9690bc963626ddea832739
|
| 10 |
+
"007": d5cbd5c920745e106b14aaf6659d95255f8b0f8d73a8b9f97ab47ecc00f72207
|
| 11 |
+
specialist_checkpoint: /home/ubuntu/minimax-external-finetune-evidence/interim678-78437a8/encoder.safetensors
|
| 12 |
+
specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 13 |
+
flow_config: /home/ubuntu/minimax-music3-release/configs/flow-encoder-v1.yaml
|
| 14 |
+
flow_config_sha256: 4a2474a50dd1e9e93c15cb5cf792e344531809a868b8f1b7405846a6012b47df
|
| 15 |
+
seed: 17090
|
| 16 |
+
negative: no clipping, no source copy, no tempo drift, no vocals
|
| 17 |
+
sample_rate: 44100
|
| 18 |
+
channels: 2
|
| 19 |
+
duration_seconds: 90.0
|
| 20 |
+
duration_tolerance_samples: 2048
|
| 21 |
+
profile_samples: 44032
|
| 22 |
+
profile_crops: 8
|
| 23 |
+
device: cpu
|
| 24 |
+
generate: false
|
| 25 |
+
train: false
|
| 26 |
+
output: /home/ubuntu/minimax-reference-ranked-pool-evidence/loveonme-exact90-f040ab1/selected.reference-ranked.wav
|
| 27 |
+
ffmpeg: /usr/bin/ffmpeg
|
configs/reference-style-v1.yaml
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.reference-style-config.v1
|
| 2 |
+
capability: reference_style_direct_latent_scorer_internal_text_bridge
|
| 3 |
+
model_id: MiniMaxAI/MiniMax-Music3
|
| 4 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
channels: 2
|
| 8 |
+
profile_samples: 44032
|
| 9 |
+
profile_maximum_crops: 8
|
| 10 |
+
default_duration_seconds: 8
|
| 11 |
+
default_candidates: 8
|
| 12 |
+
maximum_candidates: 8
|
| 13 |
+
inference_steps: 12
|
| 14 |
+
specialist_checkpoint_sha256: 3fe1f05e1d7269e3f39a89d809d17d531214cad66bd29b3c57cb6113d529ac7f
|
| 15 |
+
specialist_role: REJECTED_EXTERNAL_SPECIALIST_STYLE_SCORER_ONLY
|
| 16 |
+
native_negative_prompt_used: false
|
| 17 |
+
direct_model_audio_conditioning: false
|
| 18 |
+
full_semantic_style_transfer: false
|
| 19 |
+
artistic_quality_claim: false
|
| 20 |
+
vocal_absence_guaranteed: false
|
configs/sample-bridge-v1.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.sample-bridge-config.v1
|
| 2 |
+
capability: mir_conditioned_internal_text_bridge_not_direct_audio_embedding
|
| 3 |
+
model_id: MiniMaxAI/MiniMax-Music3
|
| 4 |
+
model_revision: fbdf52fbaaca799592917417eb05f1899f1255ec
|
| 5 |
+
diffusers_revision: 90b4e34e79a86ec5e7f2437634fe95ecd2108796
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
channels: 2
|
| 8 |
+
inference_steps: 12
|
| 9 |
+
minimum_duration_seconds: 1.0
|
| 10 |
+
maximum_duration_seconds: 30.0
|
| 11 |
+
maximum_candidates: 8
|
| 12 |
+
clipping_fraction_limit: 0.05
|
| 13 |
+
silence_rms_limit: 0.0001
|
| 14 |
+
tempo_drift_limit_octaves: 0.20
|
| 15 |
+
near_copy_similarity_limit: 0.995
|
| 16 |
+
expected_frame_tolerance: 2048
|
| 17 |
+
lyrics_literal: "[instrumental]"
|
| 18 |
+
native_negative_prompt_used: false
|
| 19 |
+
direct_audio_embedding: false
|
| 20 |
+
full_style_match: false
|
| 21 |
+
native_token_continuation: false
|
| 22 |
+
artistic_quality_claim: false
|
| 23 |
+
vocal_absence_guaranteed: false
|
configs/waveform-causal-continuation-v1.yaml
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.waveform-causal-continuation-v1-config.v1
|
| 2 |
+
capability: arbitrary_waveform_causal_short_continuation_not_native_tokens_not_semantic_style
|
| 3 |
+
seed: 260816
|
| 4 |
+
steps: 12000
|
| 5 |
+
validation_interval: 500
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
history_samples: 65536
|
| 8 |
+
prediction_samples: 8192
|
| 9 |
+
source_endpoint_guard_samples: 220500
|
| 10 |
+
model_channels: [48, 64, 96, 128, 160]
|
| 11 |
+
bottleneck_dilations: [1, 2, 4, 8, 16, 32, 64, 128]
|
| 12 |
+
residual_scale: 0.5
|
| 13 |
+
benchmark_batch_sizes_h100: [24, 32]
|
| 14 |
+
batch_24gb: 4
|
| 15 |
+
accumulation_24gb: 8
|
| 16 |
+
max_lr: 0.0002
|
| 17 |
+
min_lr: 0.00002
|
| 18 |
+
weight_decay: 0.01
|
| 19 |
+
gradient_clip_norm: 1.0
|
| 20 |
+
validation_namespace: music3lab.waveform-causal-continuation-v1.validation.v1
|
| 21 |
+
heldout_namespace: music3lab.waveform-causal-continuation-v1.heldout.v1
|
| 22 |
+
manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 23 |
+
splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
| 24 |
+
loss_weights:
|
| 25 |
+
waveform_l1: 1.0
|
| 26 |
+
waveform_l2: 0.5
|
| 27 |
+
mrstft: 0.25
|
| 28 |
+
complex_stft_nmse: 0.15
|
| 29 |
+
mid_side: 0.10
|
| 30 |
+
boundary_first1024: 0.20
|
| 31 |
+
roll_tail_samples: 1024
|
| 32 |
+
nonsilent_rms_floor: 0.0001
|
| 33 |
+
noncopy_correlation_threshold: 0.995
|
| 34 |
+
seam:
|
| 35 |
+
derivative_absolute_floor: 0.00001
|
| 36 |
+
rms_absolute_floor: 0.0001
|
| 37 |
+
maximum_median_derivative_ratio: 2.0
|
| 38 |
+
maximum_median_join_rms_log_error: 1.5
|
| 39 |
+
augmentations:
|
| 40 |
+
gain_db: [-3.0, 3.0]
|
| 41 |
+
stereo_balance_db: [-1.5, 1.5]
|
| 42 |
+
shared_polarity_probability: 0.5
|
| 43 |
+
channel_swap_probability: 0.5
|
| 44 |
+
forbidden: [side_shift, noise, resample]
|
configs/waveform-right-context-prepend-v1.yaml
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
schema_version: music3lab.waveform-right-context-prepend-v1-config.v1
|
| 2 |
+
capability: arbitrary_waveform_right_context_short_precursor_not_native_tokens_not_semantic_prepend
|
| 3 |
+
seed: 260816
|
| 4 |
+
steps: 12000
|
| 5 |
+
validation_interval: 500
|
| 6 |
+
sample_rate: 44100
|
| 7 |
+
prefix_samples: 8192
|
| 8 |
+
right_context_samples: 65536
|
| 9 |
+
source_endpoint_guard_samples: 220500
|
| 10 |
+
crop_alignment_samples: 512
|
| 11 |
+
stft: {n_fft: 1024, hop_length: 256, center: false, frames: 253}
|
| 12 |
+
encoder_channels: [48, 64, 96, 128, 160, 192]
|
| 13 |
+
transformer: {blocks: 6, width: 256, heads: 8, bidirectional: true}
|
| 14 |
+
target_queries: 29
|
| 15 |
+
benchmark_batch_sizes_h100: [24, 32]
|
| 16 |
+
max_lr: 0.0002
|
| 17 |
+
min_lr: 0.00002
|
| 18 |
+
weight_decay: 0.01
|
| 19 |
+
gradient_clip_norm: 1.0
|
| 20 |
+
validation_namespace: music3lab.waveform-right-context-prepend-v1.validation.v1
|
| 21 |
+
heldout_namespace: music3lab.waveform-right-context-prepend-v1.heldout.v1
|
| 22 |
+
manifest_sha256: 34ea6db35ff1c23aad6c8ddd6472849939ba4e19bf37e975930b441e8efc4c62
|
| 23 |
+
splits_sha256: 22aaf3186e8c1c8821b86ef7918cb63016b5821daea94622b1c024d24f66db85
|
| 24 |
+
loss_weights:
|
| 25 |
+
waveform_l1: 1.0
|
| 26 |
+
waveform_l2: 0.25
|
| 27 |
+
mrstft: 0.30
|
| 28 |
+
complex_stft_nmse: 0.20
|
| 29 |
+
mid_side: 0.10
|
| 30 |
+
seam_waveform_first_difference_last1024: 0.25
|
| 31 |
+
seam_rms_log: 0.10
|
| 32 |
+
nonsilent_rms_floor: 0.0001
|
| 33 |
+
user22_policy: ood_only_after_heldout_decision
|
| 34 |
+
augmentations:
|
| 35 |
+
gain_db: [-3.0, 3.0]
|
| 36 |
+
stereo_balance_db: [-1.5, 1.5]
|
| 37 |
+
shared_polarity_probability: 0.5
|
| 38 |
+
channel_swap_probability: 0.5
|
| 39 |
+
forbidden: [side_shift, noise, resample]
|
data/laion_disco/UPSTREAM_README.md
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
dataset_info:
|
| 4 |
+
features:
|
| 5 |
+
- name: song_id
|
| 6 |
+
dtype: string
|
| 7 |
+
- name: title
|
| 8 |
+
dtype: string
|
| 9 |
+
- name: artist_names
|
| 10 |
+
sequence: string
|
| 11 |
+
- name: artist_ids
|
| 12 |
+
sequence: string
|
| 13 |
+
- name: album_name
|
| 14 |
+
dtype: string
|
| 15 |
+
- name: album_id
|
| 16 |
+
dtype: string
|
| 17 |
+
- name: isExplicit
|
| 18 |
+
dtype: bool
|
| 19 |
+
- name: views
|
| 20 |
+
dtype: string
|
| 21 |
+
- name: duration
|
| 22 |
+
dtype: int64
|
| 23 |
+
splits:
|
| 24 |
+
- name: train
|
| 25 |
+
num_bytes: 2069255857
|
| 26 |
+
num_examples: 12320916
|
| 27 |
+
download_size: 750206954
|
| 28 |
+
dataset_size: 2069255857
|
| 29 |
+
configs:
|
| 30 |
+
- config_name: default
|
| 31 |
+
data_files:
|
| 32 |
+
- split: train
|
| 33 |
+
path: data/train-*
|
| 34 |
+
tags:
|
| 35 |
+
- music
|
| 36 |
+
pretty_name: LAION DISCO
|
| 37 |
+
size_categories:
|
| 38 |
+
- 10M<n<100M
|
| 39 |
+
---
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
The LAION-DISCO-12M dataset contains 12M links to music on YouTube, inspired by the methodology of DISCO-10M. It contains song metadata (song_id, title, artist_names, artist_ids, album_name, album_id, isExplicit, views, duration) and YouTube URL, pointing to the original song on the public web. It does not contain any original audio samples and is thus an index dataset.
|
| 43 |
+
|
| 44 |
+
Starting from an initial seed list of artists, we can discover new artists by recursively exploring the artists listed in the "Fans might also like" section.
|
| 45 |
+
We explore the related artists graph for as long as we are able to find new artists.
|
| 46 |
+
For a given artist, we can extract their metadata, such as their name and number of subscribers, as well as a list of all of their songs and music videos.
|
| 47 |
+
Importantly, each song or music video is associated with a YouTube URL (obtained from its ID). The collected metadata fields are: song_id, title, artist_names, artist_ids, album_name, album_id, isExplicit, views, duration.
|
| 48 |
+
|
| 49 |
+
The authors of DISCO-10M used a seed list of 18 artists, chosen to represent a variety of genres. However, we found that this is not sufficient for exploring the artist graph of YouTube Music. Starting from this seed list, we were able to discover only 90,007 artists and 5,399,389 songs.
|
| 50 |
+
|
| 51 |
+
We therefore compiled a larger seed list by considering the artists that appear on YouTube Music charts of top songs by country and genre playlists.
|
| 52 |
+
This resulted in an initial list of 45,218 artists. The artist graph exploration starting from this seed list resulted in 250,516 artists and 12,648,485 songs.
|
| 53 |
+
|
| 54 |
+
This work was inspired by [DISCO-10M](https://arxiv.org/abs/2306.13512), consider citing them if you use this dataset.
|
data/laion_disco/candidates.summary.json
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"candidate_manifest_sha256": "533d8623268b18f87de2774ecf45656520f37edd8d5bb30c69f9bb6401231a07",
|
| 3 |
+
"counts": {
|
| 4 |
+
"album_cap_rejected": 110,
|
| 5 |
+
"artist_cap_rejected": 163,
|
| 6 |
+
"duration_filtered": 1179263,
|
| 7 |
+
"metadata_rows": 12320916,
|
| 8 |
+
"rows_scanned": 12320916
|
| 9 |
+
},
|
| 10 |
+
"dataset": "laion/LAION-DISCO-12M",
|
| 11 |
+
"first_rank": "0000004322543062a3737fdb182a7ead063119640afd3745ffe658991041948a",
|
| 12 |
+
"last_rank": "0077fb6c207e2a35049f7547f4ec40da5dbc1a497341391903475a347b37eb8a",
|
| 13 |
+
"revision": "6e7bf3758a77301e46a715af894fefd79bb1da53",
|
| 14 |
+
"schema_version": 1,
|
| 15 |
+
"selection": {
|
| 16 |
+
"album_cap": 1,
|
| 17 |
+
"artist_cap": 2,
|
| 18 |
+
"candidate_count": 20000,
|
| 19 |
+
"cap_scope": "all listed artist IDs; normalized names when IDs absent",
|
| 20 |
+
"intended_success_count": 2000,
|
| 21 |
+
"metadata_duration_seconds_inclusive": [
|
| 22 |
+
45,
|
| 23 |
+
360
|
| 24 |
+
],
|
| 25 |
+
"pool_size": 250000,
|
| 26 |
+
"rank": "ascending sha256(revision + NUL + song_id)"
|
| 27 |
+
}
|
| 28 |
+
}
|