Download tools/export_models.py from maxffarrell/parakeet-v3-coreai: direct link, hf CLI and curl.
- Browser
- Download file 5.59 kB
-
https://huggingface.co/maxffarrell/parakeet-v3-coreai/resolve/main/tools/export_models.py
- Command line
-
hf download hf://maxffarrell/parakeet-v3-coreai/tools/export_models.py
-
curl -L -o export_models.py https://huggingface.co/maxffarrell/parakeet-v3-coreai/resolve/main/tools/export_models.py
5.59 kB
| """Apple-compatible Core AI exports, with pinned source weights and attribution. | |
| Uses Apple's BSD-3-Clause Parakeet recipe at 52c84ba874b2c57adcede08a671ce96ed1b3f433. | |
| Redux's base-3 packing is expanded exactly before FP16 conversion. This export | |
| does NOT retain Photon's 178 MB packed representation or its specialized kernels. | |
| """ | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| import hashlib | |
| import shutil | |
| import numpy as np | |
| import torch | |
| import transformers | |
| from huggingface_hub import snapshot_download, model_info | |
| from safetensors.torch import load_file | |
| import apple_parakeet_export as apple | |
| def load_source(repo, revision, directory): | |
| source = Path(snapshot_download(repo, revision=revision, local_dir=directory)) | |
| config = transformers.AutoConfig.from_pretrained(source) | |
| model = transformers.AutoModelForTDT.from_config(config) | |
| weights = load_file(source / "model.safetensors") | |
| ternary = source / "ternary.json" | |
| if ternary.exists(): | |
| spec = json.loads(ternary.read_text()) | |
| assert spec["format"] == "thrush-ternary-v2" | |
| for module in spec["quantized_modules"]: | |
| name, cols = module["name"], module["in_features"] | |
| packed = weights.pop(name + ".qweight").to(torch.int64) | |
| scales = weights.pop(name + ".scales").float() | |
| # Least-significant base-3 digit first, five elements per byte. | |
| digits = torch.stack([(packed // (3 ** i)) % 3 for i in range(5)], -1) | |
| codes = digits.flatten(1)[:, :cols] | |
| dense = (codes.float() - 1) * scales.repeat_interleave(module["group_size"], 1)[:, :cols] | |
| if module["as_conv1d"]: | |
| dense = dense.unsqueeze(-1) | |
| weights[name + ".weight"] = dense | |
| # The optional Photon VAD head is separate from the ASR architecture. | |
| omitted = [key for key in weights if key.startswith("vad_head.")] | |
| for key in omitted: | |
| del weights[key] | |
| model.load_state_dict(weights, strict=True) | |
| model.eval() | |
| processor = transformers.AutoProcessor.from_pretrained("nvidia/parakeet-tdt-0.6b-v3", revision="541d1f99c6b0c3cd0b11a95167540bb8edefd82b") | |
| if (source / "tokenizer.json").exists(): | |
| reference = json.loads(processor.tokenizer.backend_tokenizer.to_str())["model"]["vocab"] | |
| actual = json.loads((source / "tokenizer.json").read_text())["model"]["vocab"] | |
| assert reference == actual, "Source tokenizer differs from NVIDIA v3" | |
| return model, processor, omitted | |
| def main(): | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--model", choices=["v3", "ultra", "redux"], required=True) | |
| parser.add_argument("--revision", help="Immutable source revision; defaults to resolving the current source once") | |
| parser.add_argument("--output", type=Path, required=True) | |
| parser.add_argument("--sources", type=Path, required=True) | |
| args = parser.parse_args() | |
| repo = {"v3": "nvidia/parakeet-tdt-0.6b-v3", "ultra": "moondream/parakeet-ultra", "redux": "moondream/parakeet-redux"}[args.model] | |
| revision = args.revision or model_info(repo).sha | |
| model, processor, omitted = load_source(repo, revision, args.sources / args.model) | |
| # Bind the official exporter to the exact model and processor we validated. | |
| transformers.AutoModelForTDT.from_pretrained = lambda *a, **kw: model.to(dtype=kw.get("dtype", torch.float16)) | |
| transformers.AutoProcessor.from_pretrained = lambda *a, **kw: processor | |
| def metadata(graph): | |
| value = apple.AIModelAssetMetadata() | |
| value.author = "NVIDIA" if args.model == "v3" else "Moondream; based on NVIDIA Parakeet TDT v3" | |
| value.license = "CC-BY-4.0" | |
| value.model_description = f"{repo}@{revision}, {graph}; converted with Apple's Core AI Parakeet recipe by maxffarrell. No training. Redux ternary weights expanded before FP16 export." | |
| return value | |
| apple._build_aimodel_metadata = metadata | |
| apple.create_parakeet(str(args.output), repo, torch.float16, False, False, 5.0, False, apple.StreamingWindowArgs(chunk_frames=12, right_context_frames=25, left_context_frames=113)) | |
| bundle, _ = apple._bundle_paths(str(args.output), repo, torch.float16, False, {"usable_encoder_frames": 150}) | |
| provenance = {"source": repo, "revision": revision, "license": "CC-BY-4.0", "apple_recipe_revision": "52c84ba874b2c57adcede08a671ce96ed1b3f433", "precision": "float16", "omitted_auxiliary_tensors": omitted, "redux_packing_preserved": False, "streaming": True, "streaming_geometry": json.loads((bundle / "metadata.json").read_text())["streaming"]} | |
| (bundle / "provenance.json").write_text(json.dumps(provenance, indent=2) + "\n") | |
| (bundle / "NOTICE.md").write_text(f"# Attribution\n\nSource: [{repo}](https://huggingface.co/{repo}/tree/{revision}).\n\nParakeet TDT v3 is by NVIDIA. Ultra and Redux are post-trained by Moondream.\nWeights remain CC-BY-4.0: https://creativecommons.org/licenses/by/4.0/\n\nConversion by maxffarrell using Apple's BSD-3-Clause exporter. No training or endorsement. The optional Photon VAD head is not exported. Redux ternary weights are expanded then stored at FP16; the original packed size and Photon performance do not apply.\n") | |
| shutil.copyfile(Path(__file__).with_name("APPLE_LICENSE"), bundle / "APPLE_LICENSE") | |
| files = sorted(p for p in bundle.rglob("*") if p.is_file()) | |
| (bundle / "SHA256SUMS").write_text("".join(f"{hashlib.sha256(p.read_bytes()).hexdigest()} {p.relative_to(bundle)}\n" for p in files)) | |
| print("BUNDLE", bundle, flush=True) | |
| if __name__ == "__main__": | |
| torch.set_num_threads(4) | |
| main() | |