#!/usr/bin/env python3 """Export Audio8-ASR-Infinite to the WinSTT ONNX bundle. python audio8_infinite_export.py --ckpt --out --precision int4 int8 fp16 [fp32] [--decoder-shards 3] # fp32 only: split the 12 GB decoder for low-RAM parity Writes (per precision P; fp32 uses no suffix): audio_encoder{_P}.onnx + .onnx.data mel + conv + 32-layer tower + projector (one step) decoder{_P}.onnx + .onnx.data 36-layer LM step + tied LM head + semantic-VAD heads plus the precision-independent host tables: embed_tokens.bf16 raw little-endian bf16 [151936, 2048] token embedding (exact checkpoint bits) ada_scale.f32 raw f32 [len(combos), 36, 2048] per-layer (1 + AdaRMSNorm(t_cond)) runtime.json geometry, streaming clock, special ids, rolling policy, ada combo index """ from __future__ import annotations import argparse import json import shutil import time from pathlib import Path import numpy as np import audio8_infinite_ref as a8i import audio8_infinite_graph as a8i_onnx def write_tables(W, dims: a8i.Dims, cfg: dict, out: Path): arr, tag = W.raw("language_model.model.embed_tokens.weight") assert tag == "bf16" and arr.shape == (dims.vocab, dims.t_hidden), (tag, arr.shape) with open(out / "embed_tokens.bf16", "wb") as f: for s in range(0, arr.shape[0], 8192): f.write(np.ascontiguousarray(arr[s: s + 8192]).astype("