File size: 10,832 Bytes
f532e9a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
#!/usr/bin/env python3
"""One-shot: a UVR MDX-Net .onnx -> a GPU/ANE-ready fp16 CoreML .mlpackage for iOS.

Independent STFT-outside-the-graph export: the STFT/iSTFT stay OUTSIDE the graph (in the app's own
DSP), so the .mlpackage is the learned core only, with the ONNX's
NCHW I/O kept intact β€” input [1, 4, dim_f, 256] complex-as-channels [L_re, L_im, R_re, R_im] -> output
same shape = the vocal spectrogram.

Pipeline (each step gated so a bad export fails loudly, not silently):
  1. onnx2torch(onnx) -> nn.Module
  2. SNR-gate the torch module vs ONNX Runtime (must be > --min-snr, default 100 dB)
  3. torch.jit.trace -> coremltools.convert(mlprogram, fp16, compute_units, iOS target) -> .mlpackage
  4. verify: input shape [1,4,dim_f,256] from the saved spec; on macOS also a CoreML-vs-ONNX predict SNR.

Why no op-level surgery. CoreML's converter and its Metal-GPU / Apple-Neural-Engine backends lower
`ConvTranspose + ReLU` natively, so targeting `mlprogram` is a single clean convert β€” no op-version
patching, no post-conversion fixes.

fp16 (`compute_precision=FLOAT16`) is the ~2x lever and is MDX-safe (peak activation << the 65504 fp16
ceiling; MDX has no whole-tensor reduction). The Apple Neural Engine is fp16-native.

Compute units are a LOAD-TIME choice on-device (Swift `MLModelConfiguration.computeUnits`). Passing
`--compute-units` here only sets the default the Python `predict` uses on macOS; the shipped iOS app
picks the accelerator itself. Default is ALL (ANE + GPU + CPU) β€” see README.

Env (pinned, see requirements.txt): coremltools 9.0, onnx2torch 1.5.15, torch 2.9.1+cpu, onnx,
onnxruntime, numpy, Python 3.11. NOTE: converting runs anywhere; `predict()` (the numeric CoreML check
in step 4) is macOS-only β€” on Linux the .mlpackage is produced and structurally checked, and the
CoreML-vs-ONNX SNR is reported as macOS-pending.

Usage:
    python export_mdx_coreml.py UVR_MDXNET_9482.onnx     UVR_MDXNET_9482.mlpackage
    python export_mdx_coreml.py UVR-MDX-NET-Voc_FT.onnx  UVR-MDX-NET-Voc_FT.mlpackage
"""
import argparse
import os
import platform
import sys
import types
import warnings

warnings.filterwarnings("ignore")
os.environ.setdefault("GRPC_VERBOSITY", "NONE")

# onnx2torch eagerly imports torchvision; MDX uses no vision ops, so stub it out when the real package
# is missing or ABI-broken in the conversion env.
try:
    import torchvision  # noqa: F401
except Exception:
    _tv = types.ModuleType("torchvision")
    _tv.__version__ = "0.0.0"
    _ops = types.ModuleType("torchvision.ops")

    def _stub(*a, **k):
        raise NotImplementedError("torchvision op unavailable (stub)")

    for _n in ("nms", "batched_nms", "roi_align", "RoIAlign", "DeformConv2d", "deform_conv2d"):
        setattr(_ops, _n, _stub)
    _tv.ops = _ops
    sys.modules["torchvision"] = _tv
    sys.modules["torchvision.ops"] = _ops

import numpy as np
import onnx
import onnx2torch
import onnxruntime as ort
import torch
import coremltools as ct

DIM_T = 256  # native trained segment (~5.92 s @ hop 1024); static in the shipped .mlpackage

# --compute-units / --deployment-target string -> coremltools enum.
COMPUTE_UNITS = {
    "all": ct.ComputeUnit.ALL,            # ANE + GPU + CPU (default; best on Apple silicon)
    "cpuAndGPU": ct.ComputeUnit.CPU_AND_GPU,
    "cpuAndNE": ct.ComputeUnit.CPU_AND_NE,
    "cpuOnly": ct.ComputeUnit.CPU_ONLY,
}
DEPLOY_TARGETS = {
    "iOS15": ct.target.iOS15,
    "iOS16": ct.target.iOS16,
    "iOS17": ct.target.iOS17,
    "iOS18": ct.target.iOS18,
}
IN_NAME = "input"    # CaC spectrogram   [1, 4, dim_f, 256]  (Swift MLFeatureProvider key)
OUT_NAME = "output"  # vocal spectrogram [1, 4, dim_f, 256]


def onnx_dim_f(onnx_path):
    """MDX input is [batch, 4, dim_f, dim_t]; return dim_f (2048 for 9482, 3072 for Voc FT)."""
    dims = onnx.load(onnx_path).graph.input[0].type.tensor_type.shape.dim
    return int(dims[2].dim_value)


def sample(dim_f):
    return np.random.randn(1, 4, dim_f, DIM_T).astype(np.float32)


def snr(ref, got):
    den = np.sqrt((ref ** 2).mean())
    num = np.sqrt(((ref - got) ** 2).mean())
    return 20.0 * np.log10(den / (num + 1e-12))


def module_snr(mod, onnx_path, dim_f, trials=2):
    """Fidelity of the onnx2torch module vs ONNX Runtime β€” the CoreML conversion INPUT. This gate runs
    fully on Linux (no CoreML runtime needed)."""
    sess = ort.InferenceSession(onnx_path, providers=["CPUExecutionProvider"])
    iname = sess.get_inputs()[0].name
    worst = 1e9
    for _ in range(trials):
        x = sample(dim_f)
        ref = sess.run(None, {iname: x})[0]
        with torch.no_grad():
            got = mod(torch.from_numpy(x)).numpy()
        worst = min(worst, snr(ref, got))
    return float(worst)


def dir_size_mb(path):
    total = 0
    for root, _, files in os.walk(path):
        for f in files:
            total += os.path.getsize(os.path.join(root, f))
    return total / 1e6


def coreml_predict_snr(mlmodel, onnx_path, dim_f, trials=2):
    """macOS-only: run the compiled CoreML model and compare vs ONNX Runtime. Raises on non-macOS
    (no CoreML runtime), which the caller catches and reports as macOS-pending."""
    sess = ort.InferenceSession(onnx_path, providers=["CPUExecutionProvider"])
    iname = sess.get_inputs()[0].name
    worst, oshape = 1e9, None
    for _ in range(trials):
        x = sample(dim_f)                       # fp32 reference input
        ref = sess.run(None, {iname: x})[0]     # fp32 ONNX golden
        # Model I/O is fp16: feed the fp16 view so SNR reflects the true deployed path (fp16 I/O +
        # fp16 compute) vs the fp32 ONNX golden.
        got = mlmodel.predict({IN_NAME: x.astype(np.float16)})[OUT_NAME]
        oshape = tuple(np.asarray(got).shape)
        worst = min(worst, snr(ref, np.asarray(got, dtype=np.float32)))
    return float(worst), oshape


def main():
    ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
    ap.add_argument("onnx", help="source UVR MDX-Net .onnx")
    ap.add_argument("out", help="output .mlpackage")
    ap.add_argument("--min-snr", type=float, default=100.0)
    ap.add_argument("--compute-units", choices=list(COMPUTE_UNITS), default="all",
                    help="baked default for Python predict; the iOS app sets this at load time (default: all)")
    ap.add_argument("--deployment-target", choices=list(DEPLOY_TARGETS), default="iOS16")
    args = ap.parse_args()

    dim_f = onnx_dim_f(args.onnx)
    print(f"[1/4] onnx2torch: {args.onnx}  (dim_f={dim_f}, dim_t={DIM_T})", flush=True)
    mod = onnx2torch.convert(args.onnx).eval()

    msnr = module_snr(mod, args.onnx, dim_f)
    print(f"[2/4] torch-module SNR vs ONNX Runtime = {msnr:.1f} dB", flush=True)
    if msnr < args.min_snr:
        raise SystemExit(f"FAIL: onnx2torch fidelity {msnr:.1f} dB < {args.min_snr} dB (do not ship)")

    print(f"[3/4] trace + coremltools.convert -> {args.out} "
          f"(mlprogram, fp16, compute_units={args.compute_units}, {args.deployment_target})",
          flush=True)
    example = torch.from_numpy(sample(dim_f))
    with torch.no_grad():
        traced = torch.jit.trace(mod, example)
    # NCHW I/O is kept from the ONNX so the app's CaC packing feeds it directly (no transpose).
    # I/O is real fp16 (dtype `np.float16`, matching the fp16 compute) β€” the fastest, most idiomatic
    # config for an fp16 ANE/GPU model: no fp32<->fp16 boundary cast, half the I/O bandwidth on the
    # ~2M-element tensors per chunk, and the ANE's native precision end to end. It is quality-safe
    # because the pipeline normalizes the mix, so the input spectrogram sits far below the 65504 fp16
    # ceiling (peak activation ~560/1384 measured β€” fp16-safety end to end, extended to
    # the boundary). The Swift side feeds/reads an MLMultiArray(.float16); packing the fp32 host-DSP
    # spectrogram into fp16 (and back) is a trivial element-wise cast. See README "iOS integration".
    mlmodel = ct.convert(
        traced,
        inputs=[ct.TensorType(name=IN_NAME, shape=(1, 4, dim_f, DIM_T), dtype=np.float16)],
        outputs=[ct.TensorType(name=OUT_NAME, dtype=np.float16)],
        convert_to="mlprogram",
        compute_precision=ct.precision.FLOAT16,   # fp16: the ~2x lever, MDX-safe; ANE is fp16-native
        compute_units=COMPUTE_UNITS[args.compute_units],
        minimum_deployment_target=DEPLOY_TARGETS[args.deployment_target],
    )
    mlmodel.save(args.out)

    # [4/4] verify. Input shape comes straight from the saved spec (works on any OS). The numeric
    # CoreML-vs-ONNX SNR needs the CoreML runtime -> macOS only; on Linux report it as pending.
    # Read the ON-DISK artifact's spec (the source of truth for the shippable .mlpackage; the
    # in-memory convert result doesn't finalize the fp16 I/O boundary until serialized).
    spec = ct.models.MLModel(args.out, skip_model_load=True).get_spec()
    in_shape = list(spec.description.input[0].type.multiArrayType.shape)
    out_shape = list(spec.description.output[0].type.multiArrayType.shape)
    # CoreML ArrayFeatureType.ArrayDataType: FLOAT32=65568, FLOAT16=65552, DOUBLE=65600.
    io_dtype = {65568: "fp32", 65552: "fp16", 65600: "fp64"}.get(
        spec.description.input[0].type.multiArrayType.dataType, "?")
    size_mb = dir_size_mb(args.out)
    shape_ok = (in_shape == [1, 4, dim_f, DIM_T]
                and spec.WhichOneof("Type") == "mlProgram"
                and io_dtype == "fp16")

    csnr, cshape, verified = None, None, False
    try:
        csnr, cshape = coreml_predict_snr(mlmodel, args.onnx, dim_f)
        verified = True
    except Exception as e:  # non-macOS (no CoreML runtime) or predict failure
        note = f"{type(e).__name__}: {str(e).splitlines()[0][:120]}" if str(e) else type(e).__name__

    print(f"[4/4] verify: in={in_shape} out={out_shape or 'unspecified (shape-preserving U-Net)'} | "
          f"io={io_dtype} compute=fp16 | mlprogram | size={size_mb:.1f} MB", flush=True)
    if verified:
        ok = shape_ok and csnr >= args.min_snr and cshape == (1, 4, dim_f, DIM_T)
        print(f"      CoreML-vs-ONNX predict SNR = {csnr:.1f} dB  (macOS) | out={cshape}", flush=True)
        print("      " + ("PASS β€” GPU/ANE-ready fp16 .mlpackage" if ok else "FAIL β€” do not ship"), flush=True)
        raise SystemExit(0 if ok else 1)
    else:
        ok = shape_ok
        print(f"      CoreML predict SNR: MACOS-PENDING (no CoreML runtime here β€” {note})", flush=True)
        print("      re-run this script on macOS to get the on-CoreML fidelity number", flush=True)
        print("      " + ("PASS (structural) β€” .mlpackage produced; numeric CoreML check macOS-pending"
                          if ok else "FAIL β€” input shape mismatch"), flush=True)
        raise SystemExit(0 if ok else 1)


if __name__ == "__main__":
    main()