parakeet-v3-coreai / tools /export_models.py
maxffarrell's picture
Publish attributed FP16 Core AI streaming conversion with graph parity validation
8f73472 verified
Raw History Blame Contribute Delete
5.59 kB
"""Apple-compatible Core AI exports, with pinned source weights and attribution.
Uses Apple's BSD-3-Clause Parakeet recipe at 52c84ba874b2c57adcede08a671ce96ed1b3f433.
Redux's base-3 packing is expanded exactly before FP16 conversion. This export
does NOT retain Photon's 178 MB packed representation or its specialized kernels.
"""
import argparse
import json
from pathlib import Path
import hashlib
import shutil
import numpy as np
import torch
import transformers
from huggingface_hub import snapshot_download, model_info
from safetensors.torch import load_file
import apple_parakeet_export as apple
def load_source(repo, revision, directory):
source = Path(snapshot_download(repo, revision=revision, local_dir=directory))
config = transformers.AutoConfig.from_pretrained(source)
model = transformers.AutoModelForTDT.from_config(config)
weights = load_file(source / "model.safetensors")
ternary = source / "ternary.json"
if ternary.exists():
spec = json.loads(ternary.read_text())
assert spec["format"] == "thrush-ternary-v2"
for module in spec["quantized_modules"]:
name, cols = module["name"], module["in_features"]
packed = weights.pop(name + ".qweight").to(torch.int64)
scales = weights.pop(name + ".scales").float()
# Least-significant base-3 digit first, five elements per byte.
digits = torch.stack([(packed // (3 ** i)) % 3 for i in range(5)], -1)
codes = digits.flatten(1)[:, :cols]
dense = (codes.float() - 1) * scales.repeat_interleave(module["group_size"], 1)[:, :cols]
if module["as_conv1d"]:
dense = dense.unsqueeze(-1)
weights[name + ".weight"] = dense
# The optional Photon VAD head is separate from the ASR architecture.
omitted = [key for key in weights if key.startswith("vad_head.")]
for key in omitted:
del weights[key]
model.load_state_dict(weights, strict=True)
model.eval()
processor = transformers.AutoProcessor.from_pretrained("nvidia/parakeet-tdt-0.6b-v3", revision="541d1f99c6b0c3cd0b11a95167540bb8edefd82b")
if (source / "tokenizer.json").exists():
reference = json.loads(processor.tokenizer.backend_tokenizer.to_str())["model"]["vocab"]
actual = json.loads((source / "tokenizer.json").read_text())["model"]["vocab"]
assert reference == actual, "Source tokenizer differs from NVIDIA v3"
return model, processor, omitted
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--model", choices=["v3", "ultra", "redux"], required=True)
parser.add_argument("--revision", help="Immutable source revision; defaults to resolving the current source once")
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--sources", type=Path, required=True)
args = parser.parse_args()
repo = {"v3": "nvidia/parakeet-tdt-0.6b-v3", "ultra": "moondream/parakeet-ultra", "redux": "moondream/parakeet-redux"}[args.model]
revision = args.revision or model_info(repo).sha
model, processor, omitted = load_source(repo, revision, args.sources / args.model)
# Bind the official exporter to the exact model and processor we validated.
transformers.AutoModelForTDT.from_pretrained = lambda *a, **kw: model.to(dtype=kw.get("dtype", torch.float16))
transformers.AutoProcessor.from_pretrained = lambda *a, **kw: processor
def metadata(graph):
value = apple.AIModelAssetMetadata()
value.author = "NVIDIA" if args.model == "v3" else "Moondream; based on NVIDIA Parakeet TDT v3"
value.license = "CC-BY-4.0"
value.model_description = f"{repo}@{revision}, {graph}; converted with Apple's Core AI Parakeet recipe by maxffarrell. No training. Redux ternary weights expanded before FP16 export."
return value
apple._build_aimodel_metadata = metadata
apple.create_parakeet(str(args.output), repo, torch.float16, False, False, 5.0, False, apple.StreamingWindowArgs(chunk_frames=12, right_context_frames=25, left_context_frames=113))
bundle, _ = apple._bundle_paths(str(args.output), repo, torch.float16, False, {"usable_encoder_frames": 150})
provenance = {"source": repo, "revision": revision, "license": "CC-BY-4.0", "apple_recipe_revision": "52c84ba874b2c57adcede08a671ce96ed1b3f433", "precision": "float16", "omitted_auxiliary_tensors": omitted, "redux_packing_preserved": False, "streaming": True, "streaming_geometry": json.loads((bundle / "metadata.json").read_text())["streaming"]}
(bundle / "provenance.json").write_text(json.dumps(provenance, indent=2) + "\n")
(bundle / "NOTICE.md").write_text(f"# Attribution\n\nSource: [{repo}](https://huggingface.co/{repo}/tree/{revision}).\n\nParakeet TDT v3 is by NVIDIA. Ultra and Redux are post-trained by Moondream.\nWeights remain CC-BY-4.0: https://creativecommons.org/licenses/by/4.0/\n\nConversion by maxffarrell using Apple's BSD-3-Clause exporter. No training or endorsement. The optional Photon VAD head is not exported. Redux ternary weights are expanded then stored at FP16; the original packed size and Photon performance do not apply.\n")
shutil.copyfile(Path(__file__).with_name("APPLE_LICENSE"), bundle / "APPLE_LICENSE")
files = sorted(p for p in bundle.rglob("*") if p.is_file())
(bundle / "SHA256SUMS").write_text("".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()} {p.relative_to(bundle)}\n" for p in files))
print("BUNDLE", bundle, flush=True)
if __name__ == "__main__":
torch.set_num_threads(4)
main()