{
  "schema_version": 1,
  "model": "MVSep Mega 53 Stems v1",
  "author": "ZFTurbo (Roman Solovyev)",
  "license": "MIT",
  "release_url": "https://github.com/ZFTurbo/Music-Source-Separation-Training/releases/tag/v1.0.21",
  "release_tag": "v1.0.21",
  "published_at": "2026-04-20T21:29:24Z",
  "release_notes": "First version of model with 53 stems. It's based on BS Roformer architecture.\r\n\r\nList of stems: `['accordion', 'acoustic-guitar', 'back-vocal', 'banjo', 'bass', 'bassoon', 'bells', 'bowed_strings', 'brass', 'cello', 'clarinet',  'congas', 'digital-piano', 'dobro', 'double-bass', 'drums', 'electric-guitar', 'flute', 'french-horn', 'glockenspiel', 'guitar', 'harmonica', 'harp', 'harpsichord', 'hh', 'keys', 'kick', 'lead-vocal', 'mandolin', 'marimba', 'oboe', 'organ', 'percussion', 'piano', 'saxophone', 'sitar', 'snare', 'strings', 'synth', 'tambourine', 'timpani', 'toms', 'triangle', 'trombone', 'trumpet', 'tuba', 'ukulele', 'viola', 'violin', 'vocal', 'wind', 'wind-chimes',  'woodwind']`\r\n\r\nThere are some model notes:\r\n* We are continuing to refine and improve the model.\r\n* The model is memory-intensive, even with a batch size of 1. I recommend at least 16 GB of VRAM.\r\n* Currently, the model’s performance on individual stems may be lower than the specialized models available on the MVSep site.\r\n* Stems do not sum up to the original mixture, as some may contain overlapping information. For example, the \"vocals\" stem includes both lead and backing vocals.\r\n* The model returns all stems by default, but filtering out empty stems during post-processing is a logical next step.\r\n",
  "official_assets": [
    {
      "name": "mvsep_mega_model_bs_roformer_53_stems.yaml",
      "size": 4184,
      "url": "https://github.com/ZFTurbo/Music-Source-Separation-Training/releases/download/v1.0.21/mvsep_mega_model_bs_roformer_53_stems.yaml",
      "digest": "sha256:7e198062a251587088adb91215a4f44ab59e67bd62fcc805cf54d6e7dfc51103"
    },
    {
      "name": "mvsep_mega_model_bs_roformer_53_stems_v1.ckpt",
      "size": 1368919887,
      "url": "https://github.com/ZFTurbo/Music-Source-Separation-Training/releases/download/v1.0.21/mvsep_mega_model_bs_roformer_53_stems_v1.ckpt",
      "digest": "sha256:c62820893bbf86d4e734f966bd142d9157cfc8bb8e79e9d8f9ea553f3ff3519f"
    }
  ],
  "weight_license_evidence": {
    "url": "https://github.com/ZFTurbo/Music-Source-Separation-Training/issues/245#issuecomment-5834608838",
    "author": "ZFTurbo",
    "created_at": "2026-09-25T15:03:18Z",
    "full_text": "@MJMtoolsmikajuola \n\nI have released the model weights for 53-stem model under the MIT License. This license fully permits all the points you listed. However, please note that I do not hold the copyright to all audio data used to train this model. As explicitly stated in the MIT License, the software and model weights are provided \"AS IS\", WITHOUT WARRANTY OF ANY KIND. I cannot provide any legal guarantees or indemnification regarding potential claims related to the training data. You are completely free to use the model under the terms of the MIT license, but any commercial integration is done at your own risk.\n\nThe same answer is for @gabrielpamplonapg. Except I'm not author of all models on pretrained_models.md list."
  },
  "architecture_source": "https://github.com/ZFTurbo/Music-Source-Separation-Training/blob/ea7eb9c/models/bs_roformer/bs_roformer.py",
  "inference_dependency": "audio-separator 0.47.0 BSRoformer (MIT), based on lucidrains/BS-RoFormer (MIT)",
  "inference_changes": [
    "Memory-map full checkpoint; strict state dictionary load",
    "Shared backbone once per chunk, original requested heads evaluated sequentially",
    "Reference DC suppression and explicit ISTFT length",
    "Reference 20-second (882000 sample) windows by default; shorter windows configurable",
    "Float32 original, stems and residual; reference 2-way (50 percent) Hann overlap-add by default, optional 4-way or 8-way overlap",
    "Optional drum-rest is the original drums head minus selected drum child heads; it is a virtual residual, not a new trained cymbal model"
  ],
  "verified_on": "2026-10-09",
  "browser_conversion": {
    "parts": 79,
    "official_heads": 53,
    "checkpoint_values": "Original FP16 weights; no additional quantization",
    "format": "ONNX, 53 mask heads, 24 alternating time/frequency transformers, band split, final norm",
    "production_variant": "ORT com.microsoft.MultiHeadAttention with original RoPE, gates, 8 heads and scale 0.125; 20-second, 50-percent overlap retained",
    "binding_rewrite": "Hierarchical Split/Concat with at most 8 inputs or outputs; bit-exact ONNX CPU validation",
    "arithmetic_note": "Different GPU kernels can round differently; numerical equality to native CUDA is not guaranteed",
    "source_and_license": "Converted from the official MIT checkpoint and MIT architecture; complete notices in MODEL_LICENSE.txt",
    "deployed_hashes": "/browser/model-manifest.json",
    "audio_processing": "Visitor browser WebGPU/CPU DSP only; user audio is not part of the model distribution"
  }
}