Update README.md
Browse files
README.md
CHANGED
|
@@ -11,136 +11,180 @@ tags:
|
|
| 11 |
|
| 12 |
# AudioSeparatorONNX
|
| 13 |
|
| 14 |
-
|
| 15 |
|
| 16 |
-
##
|
| 17 |
|
| 18 |
-
|
| 19 |
|
| 20 |
-
|
| 21 |
-
-
|
| 22 |
-
- **
|
| 23 |
-
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
|
| 25 |
---
|
| 26 |
|
| 27 |
-
##
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
|
| 29 |
-
##
|
| 30 |
|
| 31 |
-
|
| 32 |
|
| 33 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 34 |
|
| 35 |
```python
|
| 36 |
import json
|
| 37 |
import onnxruntime as ort
|
| 38 |
|
| 39 |
-
sess = ort.InferenceSession("
|
| 40 |
-
|
| 41 |
-
|
|
|
|
|
|
|
| 42 |
|
| 43 |
-
print(sep_meta["arch"])
|
| 44 |
-
print(sep_meta["primary_stem"])
|
| 45 |
-
print(sep_meta["secondary_stem"]) # "Reverb"
|
| 46 |
```
|
| 47 |
|
| 48 |
-
###
|
| 49 |
|
| 50 |
```python
|
| 51 |
import json
|
| 52 |
import onnx
|
| 53 |
|
| 54 |
-
model = onnx.load("UVR-
|
| 55 |
-
|
| 56 |
-
|
|
|
|
|
|
|
| 57 |
```
|
| 58 |
|
| 59 |
-
###
|
| 60 |
|
| 61 |
-
|
| 62 |
|
| 63 |
```c
|
| 64 |
#include <stdio.h>
|
| 65 |
#include <stdlib.h>
|
| 66 |
#include <string.h>
|
| 67 |
|
| 68 |
-
/*
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 72 |
if (!f) return NULL;
|
| 73 |
|
| 74 |
-
|
| 75 |
-
const size_t SCAN_SIZE = 64 * 1024;
|
| 76 |
fseek(f, 0, SEEK_END);
|
| 77 |
-
long
|
| 78 |
-
size_t
|
| 79 |
-
|
| 80 |
-
fseek(f, offset, SEEK_SET);
|
| 81 |
|
| 82 |
-
char* buf = (char*)malloc(
|
| 83 |
if (!buf) { fclose(f); return NULL; }
|
| 84 |
-
size_t n = fread(buf, 1,
|
| 85 |
fclose(f);
|
| 86 |
buf[n] = '\0';
|
| 87 |
|
| 88 |
-
|
| 89 |
-
const char* marker = "sep_meta";
|
| 90 |
-
char* pos = buf;
|
| 91 |
char* found = NULL;
|
| 92 |
-
|
| 93 |
-
if ((
|
| 94 |
-
memcmp(pos, marker, strlen(marker)) == 0) {
|
| 95 |
-
found = pos; /* keep the last occurrence */
|
| 96 |
-
}
|
| 97 |
-
pos++;
|
| 98 |
-
}
|
| 99 |
if (!found) { free(buf); return NULL; }
|
| 100 |
|
| 101 |
-
|
| 102 |
-
|
| 103 |
-
|
| 104 |
-
|
| 105 |
-
if (json_start >= buf + n) { free(buf); return NULL; }
|
| 106 |
|
| 107 |
-
/* Find matching closing '}' */
|
| 108 |
int depth = 0;
|
| 109 |
-
char* p = json_start;
|
| 110 |
while (p < buf + n) {
|
| 111 |
-
if
|
| 112 |
-
else if (*p == '}') {
|
| 113 |
p++;
|
| 114 |
}
|
| 115 |
if (depth != 0) { free(buf); return NULL; }
|
| 116 |
|
| 117 |
-
size_t
|
| 118 |
-
char*
|
| 119 |
-
memcpy(
|
| 120 |
-
|
| 121 |
-
|
| 122 |
free(buf);
|
| 123 |
-
return
|
| 124 |
}
|
| 125 |
|
| 126 |
-
/* Example usage */
|
| 127 |
int main(void) {
|
| 128 |
-
char*
|
| 129 |
-
|
| 130 |
-
|
| 131 |
-
|
| 132 |
-
}
|
| 133 |
return 0;
|
| 134 |
}
|
| 135 |
```
|
| 136 |
|
| 137 |
---
|
| 138 |
|
| 139 |
-
##
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 140 |
|
| 141 |
-
|
| 142 |
|
| 143 |
-
### MDX-Net
|
| 144 |
|
| 145 |
```json
|
| 146 |
{
|
|
@@ -157,20 +201,9 @@ The `sep_meta` value is a compact JSON object. Fields differ by architecture:
|
|
| 157 |
}
|
| 158 |
```
|
| 159 |
|
| 160 |
-
|
| 161 |
-
|-------|------|-------------|
|
| 162 |
-
| `n_fft` | int | STFT window size |
|
| 163 |
-
| `hop_length` | int | STFT hop length |
|
| 164 |
-
| `dim_f` | int | Frequency bins fed to the network |
|
| 165 |
-
| `dim_t` | int | Time frames per chunk |
|
| 166 |
-
| `compensate` | float | Amplitude gain applied after iSTFT |
|
| 167 |
-
| `overlap` | float | Overlap ratio between consecutive chunks (0β0.75) |
|
| 168 |
|
| 169 |
-
|
| 170 |
-
- Input: `input` β shape `(1, 4, dim_f, dim_t)` β four channels: `[real_L, imag_L, real_R, imag_R]`
|
| 171 |
-
- Output: `output` β same shape as input
|
| 172 |
-
|
| 173 |
-
### VR Architecture models
|
| 174 |
|
| 175 |
```json
|
| 176 |
{
|
|
@@ -182,217 +215,121 @@ The `sep_meta` value is a compact JSON object. Fields differ by architecture:
|
|
| 182 |
"bins": 672,
|
| 183 |
"window_size": 512,
|
| 184 |
"is_vr51": true,
|
| 185 |
-
"nn_arch_size":
|
| 186 |
"model_capacity": [32, 128],
|
| 187 |
"band_params": {
|
| 188 |
-
"1": {"sr": 11025, "
|
| 189 |
-
"2": {"sr": 22050, "
|
| 190 |
-
"3": {"sr": 44100, "
|
| 191 |
-
"4": {"sr": 44100, "
|
| 192 |
}
|
| 193 |
}
|
| 194 |
```
|
| 195 |
|
| 196 |
-
|
| 197 |
-
|-------|------|-------------|
|
| 198 |
-
| `vr_model_param` | string | Band config preset name |
|
| 199 |
-
| `bins` | int | Total frequency bins after multi-band STFT |
|
| 200 |
-
| `window_size` | int | Window size for the primary STFT pass |
|
| 201 |
-
| `is_vr51` | bool | `true` = CascadedNet (UVR v5.1), `false` = v4 |
|
| 202 |
-
| `nn_arch_size` | int | Network depth parameter |
|
| 203 |
-
| `band_params` | object | Per-band STFT config used to build the input spectrogram |
|
| 204 |
|
| 205 |
-
|
| 206 |
-
- Input: `input` β shape `(batch, 2, bins, frames)` β stereo multi-band magnitude spectrogram
|
| 207 |
-
- Output: `output` β shape `(batch, 2, bins, frames)` β predicted source mask
|
| 208 |
|
| 209 |
-
|
| 210 |
|
| 211 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 212 |
|
| 213 |
```json
|
| 214 |
{
|
| 215 |
"arch": "ROFORMER",
|
| 216 |
-
"roformer_type": "
|
| 217 |
"primary_stem": "No Reverb",
|
| 218 |
"secondary_stem": "Reverb",
|
| 219 |
"sample_rate": 44100,
|
| 220 |
-
"
|
| 221 |
-
"
|
|
|
|
|
|
|
| 222 |
"n_fft": 2048,
|
| 223 |
-
"dim_t": 256,
|
| 224 |
-
"num_bands": 64,
|
| 225 |
-
"dim": 128,
|
| 226 |
"frames": 256,
|
| 227 |
-
"n_freq_indices":
|
| 228 |
-
"
|
| 229 |
-
"n_full_freqs": 1025,
|
| 230 |
"overlap": 2,
|
| 231 |
-
"freq_indices": [...],
|
| 232 |
-
"num_bands_per_freq": [...]
|
| 233 |
}
|
| 234 |
```
|
| 235 |
|
| 236 |
-
| Field |
|
| 237 |
-
|-------|------
|
| 238 |
-
| `chunk_size` |
|
| 239 |
-
| `
|
| 240 |
-
| `
|
| 241 |
-
| `
|
| 242 |
-
| `
|
| 243 |
-
| `freq_indices` | int[] | Frequency bin indices to gather from full STFT before passing to network |
|
| 244 |
-
| `num_bands_per_freq` | int[] | Band assignment per frequency bin (for scatter-add in iSTFT) |
|
| 245 |
-
| `overlap` | int | Number of overlapping inference passes per chunk |
|
| 246 |
-
|
| 247 |
-
**ONNX I/O:**
|
| 248 |
-
- Input: `x_bands` β shape `(1, frames, num_bands, dim)` β band-split STFT features
|
| 249 |
-
- Output: `masks` β shape `(1, 1, n_freq_indices, frames, 2)` β complex mask
|
| 250 |
-
|
| 251 |
-
---
|
| 252 |
-
|
| 253 |
-
## π Quickstart with `audio-separator`
|
| 254 |
|
| 255 |
-
|
| 256 |
|
| 257 |
-
|
| 258 |
|
| 259 |
-
##
|
| 260 |
|
| 261 |
```bash
|
| 262 |
-
# CPU
|
| 263 |
-
pip install "audio-separator[
|
| 264 |
-
|
| 265 |
-
# Nvidia GPU (CUDA)
|
| 266 |
-
pip install "audio-separator[gpu]"
|
| 267 |
-
|
| 268 |
-
# Apple Silicon (CoreML acceleration)
|
| 269 |
-
pip install "audio-separator[cpu]" # CoreML is auto-detected on macOS
|
| 270 |
```
|
| 271 |
|
| 272 |
-
### CLI usage
|
| 273 |
-
|
| 274 |
-
Download the model file from the **Files and versions** tab, then point `--model_file_dir` at the folder containing it:
|
| 275 |
-
|
| 276 |
```bash
|
| 277 |
-
#
|
| 278 |
-
audio-separator mix.wav \
|
| 279 |
-
--model_filename Reverb_HQ_By_FoxJoy.onnx \
|
| 280 |
-
--model_file_dir /path/to/downloaded/models \
|
| 281 |
-
--output_dir ./output \
|
| 282 |
-
--output_format WAV
|
| 283 |
-
|
| 284 |
-
# Vocal / instrumental separation
|
| 285 |
audio-separator mix.wav \
|
| 286 |
-
--model_filename UVR-
|
| 287 |
-
--model_file_dir /path/to/
|
| 288 |
--output_dir ./output
|
| 289 |
```
|
| 290 |
|
| 291 |
-
### Python API
|
| 292 |
-
|
| 293 |
```python
|
|
|
|
| 294 |
from audio_separator.separator import Separator
|
| 295 |
|
| 296 |
-
|
| 297 |
-
|
| 298 |
-
|
| 299 |
-
output_format="WAV",
|
| 300 |
-
)
|
| 301 |
-
|
| 302 |
-
# Load and run β arch is auto-detected from the file extension
|
| 303 |
-
separator.load_model("Reverb_HQ_By_FoxJoy.onnx")
|
| 304 |
-
output_files = separator.separate("mix.wav")
|
| 305 |
-
print("Output:", output_files)
|
| 306 |
```
|
| 307 |
|
| 308 |
-
|
| 309 |
-
|
| 310 |
-
`audio-separator` exposes the most important per-arch parameters:
|
| 311 |
|
| 312 |
```python
|
| 313 |
-
|
| 314 |
-
|
| 315 |
-
|
| 316 |
-
mdx_params={
|
| 317 |
-
"hop_length": 1024,
|
| 318 |
-
"segment_size": 256, # dim_t from sep_meta
|
| 319 |
-
"overlap": 0.25, # matches sep_meta default
|
| 320 |
-
"batch_size": 1,
|
| 321 |
-
"enable_denoise": False,
|
| 322 |
-
}
|
| 323 |
)
|
| 324 |
-
|
| 325 |
-
# VR β tune aggression and window size
|
| 326 |
-
separator = Separator(
|
| 327 |
-
model_file_dir="/path/to/downloaded/models",
|
| 328 |
-
vr_params={
|
| 329 |
-
"batch_size": 1,
|
| 330 |
-
"window_size": 512, # 320 = slower but better
|
| 331 |
-
"aggression": 5, # 0-100, higher = more aggressive separation
|
| 332 |
-
"enable_tta": False,
|
| 333 |
-
"high_end_process": False,
|
| 334 |
-
}
|
| 335 |
-
)
|
| 336 |
-
|
| 337 |
-
separator.load_model("UVR-DeEcho-DeReverb.onnx")
|
| 338 |
-
output_files = separator.separate("mix.wav")
|
| 339 |
-
```
|
| 340 |
-
|
| 341 |
-
> **Note on `sep_meta`:** `audio-separator` reads model parameters from its own internal `mdx_model_data.json` lookup table β it does **not** read the embedded `sep_meta` metadata. The `sep_meta` field is provided for custom inference pipelines and other runtimes (see sections below). embed_sep_meta(onnx_path: str, meta: dict) -> None:
|
| 342 |
-
```python
|
| 343 |
-
model = onnx.load(onnx_path)
|
| 344 |
-
# Remove stale entry if present
|
| 345 |
-
existing = [p for p in model.metadata_props if p.key != "sep_meta"]
|
| 346 |
-
del model.metadata_props[:]
|
| 347 |
-
model.metadata_props.extend(existing)
|
| 348 |
-
# Add new entry
|
| 349 |
-
entry = model.metadata_props.add()
|
| 350 |
-
entry.key = "sep_meta"
|
| 351 |
-
entry.value = json.dumps(meta, separators=(",", ":"))
|
| 352 |
-
onnx.save_model(model, onnx_path)
|
| 353 |
```
|
| 354 |
-
# Example β MDX model
|
| 355 |
|
| 356 |
-
```python
|
| 357 |
-
embed_sep_meta("my_model.onnx", {
|
| 358 |
-
"arch": "MDX",
|
| 359 |
-
"primary_stem": "Vocals",
|
| 360 |
-
"secondary_stem": "Instrumental",
|
| 361 |
-
"sample_rate": 44100,
|
| 362 |
-
"n_fft": 7680,
|
| 363 |
-
"hop_length": 1024,
|
| 364 |
-
"dim_f": 3072,
|
| 365 |
-
"dim_t": 256,
|
| 366 |
-
"compensate": 1.021,
|
| 367 |
-
"overlap": 0.25,
|
| 368 |
-
})
|
| 369 |
-
```
|
| 370 |
---
|
| 371 |
|
| 372 |
## π Requirements
|
| 373 |
|
| 374 |
```
|
| 375 |
onnxruntime >= 1.16
|
| 376 |
-
numpy >= 1.24
|
| 377 |
-
soundfile >= 0.12
|
| 378 |
```
|
| 379 |
|
| 380 |
-
For metadata
|
| 381 |
```
|
| 382 |
onnx >= 1.14
|
| 383 |
```
|
| 384 |
|
|
|
|
|
|
|
| 385 |
## π License
|
| 386 |
|
| 387 |
-
|
| 388 |
-
*Please check the original model licenses before using them for commercial purposes.*
|
| 389 |
|
| 390 |
---
|
| 391 |
|
| 392 |
-
## π Acknowledgments
|
| 393 |
-
|
| 394 |
-
Special thanks to the authors and maintainers of:
|
| 395 |
|
| 396 |
-
|
| 397 |
-
|
| 398 |
-
|
|
|
|
|
|
| 11 |
|
| 12 |
# AudioSeparatorONNX
|
| 13 |
|
| 14 |
+
Popular audio source separation models β UVR5, MDX-Net, VR Architecture, BSRoformer β converted to **ONNX** format and packaged as single self-contained files.
|
| 15 |
|
| 16 |
+
## Why ONNX?
|
| 17 |
|
| 18 |
+
The standard way to run these models is through [audio-separator](https://github.com/nomadkaraoke/python-audio-separator) or [UVR](https://github.com/Anjok07/ultimatevocalremovergui), both of which require Python and PyTorch. That's fine for desktop use, but becomes a problem when you want to:
|
| 19 |
|
| 20 |
+
- Ship a **native application** (C++, Swift, Rust, .NET) without a Python runtime
|
| 21 |
+
- Run separation on a **mobile or embedded device**
|
| 22 |
+
- Build a **server-side pipeline** where spinning up PyTorch per request is too heavy
|
| 23 |
+
- Use **CoreML, TensorRT, DirectML, or other hardware accelerators** via ONNX Runtime
|
| 24 |
+
- Load a model in any language that has an ONNX Runtime binding
|
| 25 |
+
|
| 26 |
+
ONNX Runtime handles all of the above with a single lightweight library and no Python dependency. The models in this repo are ready to drop into any ORT-based pipeline.
|
| 27 |
+
|
| 28 |
+
## What makes this repo different
|
| 29 |
+
|
| 30 |
+
Most ONNX model repos ship the file and nothing else. Every model here includes two metadata blobs embedded directly inside the `.onnx` file:
|
| 31 |
+
|
| 32 |
+
| Key | What it contains |
|
| 33 |
+
|-----|-----------------|
|
| 34 |
+
| `sep_meta` | All inference parameters: arch, stems, STFT config, chunk size, overlap, and for Roformer β `freq_indices` and `num_bands_per_freq` arrays needed for the gather/scatter steps |
|
| 35 |
+
| `model_config` | The original training config reconstructed from the model weights β lets you recover a YAML for audio-separator or any other PyTorch pipeline without needing the original sidecar file |
|
| 36 |
+
|
| 37 |
+
No JSON files, no YAML sidecars, no download_checks.json. One file per model.
|
| 38 |
|
| 39 |
---
|
| 40 |
|
| 41 |
+
## β οΈ Compatibility Notes
|
| 42 |
+
|
| 43 |
+
### MDX and VR models
|
| 44 |
+
|
| 45 |
+
These work with audio-separator and UVR out of the box. Just point `model_file_dir` at the folder containing the `.onnx` files.
|
| 46 |
+
|
| 47 |
+
> **Hash detection:** audio-separator identifies models by MD5 of the last 10 MB of the file. Because `sep_meta` and `model_config` are appended at the end, the hash of these files differs from the originals in the UVR database. If auto-detection fails, pass the parameters explicitly via `mdx_params` or `vr_params` β all values are available in `sep_meta`.
|
| 48 |
|
| 49 |
+
### Roformer models (BSRoformer)
|
| 50 |
|
| 51 |
+
Roformer `.onnx` files are intended for **ONNX Runtime inference only** β audio-separator and UVR run Roformer via PyTorch from the original `.ckpt`, not from ONNX. These files are useful if you are building a custom native pipeline.
|
| 52 |
|
| 53 |
+
The embedded `model_config` key lets you recover the training configuration without the original `.yaml` sidecar β see the extraction section below.
|
| 54 |
+
|
| 55 |
+
---
|
| 56 |
+
|
| 57 |
+
## π¦ Reading Embedded Metadata
|
| 58 |
+
|
| 59 |
+
Both keys are stored as compact JSON strings inside the ONNX `metadata_props` field.
|
| 60 |
+
|
| 61 |
+
### Python β `onnxruntime`
|
| 62 |
|
| 63 |
```python
|
| 64 |
import json
|
| 65 |
import onnxruntime as ort
|
| 66 |
|
| 67 |
+
sess = ort.InferenceSession("UVR-DeEcho-DeReverb.onnx", providers=["CPUExecutionProvider"])
|
| 68 |
+
meta = sess.get_modelmeta().custom_metadata_map # dict[str, str]
|
| 69 |
+
|
| 70 |
+
sep_meta = json.loads(meta["sep_meta"])
|
| 71 |
+
model_config = json.loads(meta["model_config"])
|
| 72 |
|
| 73 |
+
print(sep_meta["arch"]) # "MDX" | "VR" | "ROFORMER"
|
| 74 |
+
print(sep_meta["primary_stem"]) # e.g. "No Reverb"
|
|
|
|
| 75 |
```
|
| 76 |
|
| 77 |
+
### Python β `onnx` library (no inference session)
|
| 78 |
|
| 79 |
```python
|
| 80 |
import json
|
| 81 |
import onnx
|
| 82 |
|
| 83 |
+
model = onnx.load("UVR-DeEcho-DeReverb.onnx")
|
| 84 |
+
meta = {p.key: p.value for p in model.metadata_props}
|
| 85 |
+
|
| 86 |
+
sep_meta = json.loads(meta["sep_meta"])
|
| 87 |
+
model_config = json.loads(meta["model_config"])
|
| 88 |
```
|
| 89 |
|
| 90 |
+
### C β byte-scan (no ORT, no Python)
|
| 91 |
|
| 92 |
+
Both keys are stored near the **end** of the ONNX protobuf, after all weight tensors. You can extract either by scanning the last 64 KB without loading any weights:
|
| 93 |
|
| 94 |
```c
|
| 95 |
#include <stdio.h>
|
| 96 |
#include <stdlib.h>
|
| 97 |
#include <string.h>
|
| 98 |
|
| 99 |
+
/*
|
| 100 |
+
* Returns a heap-allocated null-terminated JSON string for the given key,
|
| 101 |
+
* or NULL if not found. Caller must free() the result.
|
| 102 |
+
*
|
| 103 |
+
* Increase SCAN_SIZE to 524288 for large Roformer models whose
|
| 104 |
+
* freq_indices array may exceed 64 KB.
|
| 105 |
+
*/
|
| 106 |
+
char* read_onnx_meta_key(const char* path, const char* key) {
|
| 107 |
+
FILE* f = fopen(path, "rb");
|
| 108 |
if (!f) return NULL;
|
| 109 |
|
| 110 |
+
const size_t SCAN_SIZE = 65536;
|
|
|
|
| 111 |
fseek(f, 0, SEEK_END);
|
| 112 |
+
long sz = ftell(f);
|
| 113 |
+
size_t read_sz = (sz < (long)SCAN_SIZE) ? (size_t)sz : SCAN_SIZE;
|
| 114 |
+
fseek(f, sz - (long)read_sz, SEEK_SET);
|
|
|
|
| 115 |
|
| 116 |
+
char* buf = (char*)malloc(read_sz + 1);
|
| 117 |
if (!buf) { fclose(f); return NULL; }
|
| 118 |
+
size_t n = fread(buf, 1, read_sz, f);
|
| 119 |
fclose(f);
|
| 120 |
buf[n] = '\0';
|
| 121 |
|
| 122 |
+
size_t klen = strlen(key);
|
|
|
|
|
|
|
| 123 |
char* found = NULL;
|
| 124 |
+
for (size_t i = 0; i + klen < n; i++)
|
| 125 |
+
if (memcmp(buf + i, key, klen) == 0) found = buf + i;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 126 |
if (!found) { free(buf); return NULL; }
|
| 127 |
|
| 128 |
+
char* p = found + klen;
|
| 129 |
+
while (p < buf + n && *p != '{') p++;
|
| 130 |
+
if (p >= buf + n) { free(buf); return NULL; }
|
| 131 |
+
char* start = p;
|
|
|
|
| 132 |
|
|
|
|
| 133 |
int depth = 0;
|
|
|
|
| 134 |
while (p < buf + n) {
|
| 135 |
+
if (*p == '{') depth++;
|
| 136 |
+
else if (*p == '}') { if (--depth == 0) break; }
|
| 137 |
p++;
|
| 138 |
}
|
| 139 |
if (depth != 0) { free(buf); return NULL; }
|
| 140 |
|
| 141 |
+
size_t len = (size_t)(p - start) + 1;
|
| 142 |
+
char* out = (char*)malloc(len + 1);
|
| 143 |
+
memcpy(out, start, len);
|
| 144 |
+
out[len] = '\0';
|
|
|
|
| 145 |
free(buf);
|
| 146 |
+
return out;
|
| 147 |
}
|
| 148 |
|
|
|
|
| 149 |
int main(void) {
|
| 150 |
+
char* sep = read_onnx_meta_key("model.onnx", "sep_meta");
|
| 151 |
+
char* cfg = read_onnx_meta_key("model.onnx", "model_config");
|
| 152 |
+
if (sep) { printf("sep_meta: %s\n", sep); free(sep); }
|
| 153 |
+
if (cfg) { printf("model_config: %s\n", cfg); free(cfg); }
|
|
|
|
| 154 |
return 0;
|
| 155 |
}
|
| 156 |
```
|
| 157 |
|
| 158 |
---
|
| 159 |
|
| 160 |
+
## π§ Using `model_config` in your ONNX pipeline
|
| 161 |
+
|
| 162 |
+
The `model_config` key contains the original training configuration. If you are building a custom ONNX Runtime pipeline around Roformer inference, you can read it directly at runtime instead of shipping a separate YAML:
|
| 163 |
+
|
| 164 |
+
```python
|
| 165 |
+
import json
|
| 166 |
+
import onnxruntime as ort
|
| 167 |
+
|
| 168 |
+
sess = ort.InferenceSession("deverb_bs_roformer_8_384dim_10depth.onnx",
|
| 169 |
+
providers=["CPUExecutionProvider"])
|
| 170 |
+
meta = sess.get_modelmeta().custom_metadata_map
|
| 171 |
+
|
| 172 |
+
sep_meta = json.loads(meta["sep_meta"]) # chunk sizes, freq_indices, etc.
|
| 173 |
+
model_config = json.loads(meta["model_config"]) # arch params: dim, depth, n_fft, ...
|
| 174 |
+
|
| 175 |
+
# Everything you need to run the pipeline is in these two dicts.
|
| 176 |
+
# No separate YAML or JSON sidecar required.
|
| 177 |
+
n_fft = sep_meta["n_fft"]
|
| 178 |
+
hop_length = sep_meta["hop_length"]
|
| 179 |
+
chunk_size = sep_meta["chunk_size"]
|
| 180 |
+
overlap = sep_meta["overlap"]
|
| 181 |
+
```
|
| 182 |
+
|
| 183 |
+
---
|
| 184 |
|
| 185 |
+
## π¬ `sep_meta` Reference
|
| 186 |
|
| 187 |
+
### MDX-Net
|
| 188 |
|
| 189 |
```json
|
| 190 |
{
|
|
|
|
| 201 |
}
|
| 202 |
```
|
| 203 |
|
| 204 |
+
**ONNX I/O** β Input `input (1, 4, dim_f, dim_t)`: `[real_L, imag_L, real_R, imag_R]` Β· Output `output` same shape
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 205 |
|
| 206 |
+
### VR Architecture
|
|
|
|
|
|
|
|
|
|
|
|
|
| 207 |
|
| 208 |
```json
|
| 209 |
{
|
|
|
|
| 215 |
"bins": 672,
|
| 216 |
"window_size": 512,
|
| 217 |
"is_vr51": true,
|
| 218 |
+
"nn_arch_size": 218409,
|
| 219 |
"model_capacity": [32, 128],
|
| 220 |
"band_params": {
|
| 221 |
+
"1": {"sr": 11025, "hl": 480, "n_fft": 960, "crop_start": 0, "crop_stop": 245},
|
| 222 |
+
"2": {"sr": 22050, "hl": 480, "n_fft": 1920, "crop_start": 245, "crop_stop": 432},
|
| 223 |
+
"3": {"sr": 44100, "hl": 480, "n_fft": 3840, "crop_start": 432, "crop_stop": 567},
|
| 224 |
+
"4": {"sr": 44100, "hl": 960, "n_fft": 7680, "crop_start": 567, "crop_stop": 673}
|
| 225 |
}
|
| 226 |
}
|
| 227 |
```
|
| 228 |
|
| 229 |
+
**ONNX I/O** β Input `input (1, 2, bins+1, window_size)`: stereo multi-band magnitude Β· Output `output` same shape (source mask)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 230 |
|
| 231 |
+
### BSRoformer
|
|
|
|
|
|
|
| 232 |
|
| 233 |
+
The ONNX graph covers `band_split + transformer + mask_estimators`. STFT and iSTFT are handled by the caller.
|
| 234 |
|
| 235 |
+
Pipeline:
|
| 236 |
+
1. STFT per channel β interleave channels β `stft_repr` shape `(n_full_freqs, T, 2)`
|
| 237 |
+
2. Gather `freq_indices` from `stft_repr` β flatten β `x_flat` shape `(1, frames, n_freq_indices*2)` β **ONNX input**
|
| 238 |
+
3. ONNX forward β `masks` shape `(1, 1, n_freq_indices, frames, 2)`
|
| 239 |
+
4. `scatter_add` masks back to `stft_repr` positions, divide by `num_bands_per_freq` β `masks_avg`
|
| 240 |
+
5. Multiply `stft_repr * masks_avg` β iSTFT per channel β audio
|
| 241 |
|
| 242 |
```json
|
| 243 |
{
|
| 244 |
"arch": "ROFORMER",
|
| 245 |
+
"roformer_type": "BSRoformer",
|
| 246 |
"primary_stem": "No Reverb",
|
| 247 |
"secondary_stem": "Reverb",
|
| 248 |
"sample_rate": 44100,
|
| 249 |
+
"num_channels": 2,
|
| 250 |
+
"chunk_size": 112455,
|
| 251 |
+
"native_chunk_size": 352800,
|
| 252 |
+
"hop_length": 441,
|
| 253 |
"n_fft": 2048,
|
|
|
|
|
|
|
|
|
|
| 254 |
"frames": 256,
|
| 255 |
+
"n_freq_indices": 2050,
|
| 256 |
+
"n_full_freqs": 2050,
|
|
|
|
| 257 |
"overlap": 2,
|
| 258 |
+
"freq_indices": [0, 1, 2, "..."],
|
| 259 |
+
"num_bands_per_freq": [1, 1, 1, "..."]
|
| 260 |
}
|
| 261 |
```
|
| 262 |
|
| 263 |
+
| Field | Description |
|
| 264 |
+
|-------|-------------|
|
| 265 |
+
| `chunk_size` | Samples per inference chunk (export size β smaller to reduce RAM during export) |
|
| 266 |
+
| `native_chunk_size` | Original training chunk size β use for best quality if RAM allows |
|
| 267 |
+
| `frames` | STFT frame count β `x_flat` must have exactly this many time frames |
|
| 268 |
+
| `freq_indices` | Indices into `stft_repr` to gather before the ONNX forward pass |
|
| 269 |
+
| `num_bands_per_freq` | How many bands cover each frequency bin β scatter normalization denominator |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 270 |
|
| 271 |
+
**ONNX I/O** β Input `x_flat (1, frames, n_freq_indices*2)` Β· Output `masks (1, 1, n_freq_indices, frames, 2)`
|
| 272 |
|
| 273 |
+
---
|
| 274 |
|
| 275 |
+
## π Quickstart β MDX and VR with `audio-separator`
|
| 276 |
|
| 277 |
```bash
|
| 278 |
+
pip install "audio-separator[cpu]" # CPU / Apple Silicon
|
| 279 |
+
pip install "audio-separator[gpu]" # Nvidia CUDA
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 280 |
```
|
| 281 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 282 |
```bash
|
| 283 |
+
# CLI
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 284 |
audio-separator mix.wav \
|
| 285 |
+
--model_filename UVR-DeEcho-DeReverb.onnx \
|
| 286 |
+
--model_file_dir /path/to/models \
|
| 287 |
--output_dir ./output
|
| 288 |
```
|
| 289 |
|
|
|
|
|
|
|
| 290 |
```python
|
| 291 |
+
# Python API
|
| 292 |
from audio_separator.separator import Separator
|
| 293 |
|
| 294 |
+
sep = Separator(model_file_dir="/path/to/models", output_dir="./output")
|
| 295 |
+
sep.load_model("UVR-DeEcho-DeReverb.onnx")
|
| 296 |
+
sep.separate("mix.wav")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 297 |
```
|
| 298 |
|
| 299 |
+
If hash auto-detection fails (see compatibility note above), pass the parameters manually:
|
|
|
|
|
|
|
| 300 |
|
| 301 |
```python
|
| 302 |
+
sep = Separator(
|
| 303 |
+
model_file_dir="/path/to/models",
|
| 304 |
+
mdx_params={"hop_length": 1024, "segment_size": 256, "overlap": 0.25},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 305 |
)
|
| 306 |
+
sep.load_model("UVR-MDX-NET-Inst_HQ_5.onnx")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 307 |
```
|
|
|
|
| 308 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 309 |
---
|
| 310 |
|
| 311 |
## π Requirements
|
| 312 |
|
| 313 |
```
|
| 314 |
onnxruntime >= 1.16
|
|
|
|
|
|
|
| 315 |
```
|
| 316 |
|
| 317 |
+
For reading metadata without running inference:
|
| 318 |
```
|
| 319 |
onnx >= 1.14
|
| 320 |
```
|
| 321 |
|
| 322 |
+
---
|
| 323 |
+
|
| 324 |
## π License
|
| 325 |
|
| 326 |
+
[MIT License](LICENSE). Check individual model licenses before commercial use.
|
|
|
|
| 327 |
|
| 328 |
---
|
| 329 |
|
| 330 |
+
## π Acknowledgments
|
|
|
|
|
|
|
| 331 |
|
| 332 |
+
- [Ultimate Vocal Remover (UVR5)](https://github.com/Anjok07/ultimatevocalremovergui)
|
| 333 |
+
- [audio-separator](https://github.com/nomadkaraoke/python-audio-separator)
|
| 334 |
+
- [ONNX Runtime](https://onnxruntime.ai/)
|
| 335 |
+
- [BS-RoFormer](https://github.com/lucidrains/BS-RoFormer)
|