File size: 8,328 Bytes
29f25be | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 | """Mac-only Core ML conversion/compression/runtime, imported on demand."""
import platform
import time
from collections import Counter
from pathlib import Path
import numpy as np
import torch
from vimeml.deployment.bundle import (BundleLM, environment, fresh_directory, read_json,
tree_inventory, write_json)
from vimeml.deployment.graph import trace_graph
from vimeml.training.data import file_sha
def coremltools():
if platform.system() != "Darwin":
raise RuntimeError("Run Core ML conversion, compression and prediction manually on Mac.")
import coremltools as ct
return ct
def inspect_spec(model):
spec = model.get_spec()
operations = Counter()
def walk(block):
for op in block.operations:
operations[op.type] += 1
for child in op.blocks:
walk(child)
for function in spec.mlProgram.functions.values():
for block in function.block_specializations.values():
walk(block)
return {"specification_version": spec.specificationVersion, "operations": dict(operations),
"inputs": [f.name for f in spec.description.input], "outputs": [f.name for f in spec.description.output]}
def finish(output, model, metadata):
fresh_directory(output)
package = output / "model.mlpackage"
model.save(str(package))
inventory = tree_inventory(package)
write_json(output / "manifest.json", {"format": "vimeml_coreml_v1", "status": "complete", **metadata,
"environment": environment(), "spec": inspect_spec(model), "package_files": inventory,
"package_bytes": sum(entry["bytes"] for entry in inventory.values()),
"note": "Logical file bytes, not compiled size, IPA size or resident memory. Inspect weight.bin and const/constexpr ops for tied-weight duplication."})
def verify_package(directory):
directory = Path(directory)
manifest = read_json(directory / "manifest.json")
if manifest.get("format") != "vimeml_coreml_v1" or manifest.get("status") != "complete":
raise ValueError("Incomplete/unsupported Core ML experiment.")
if tree_inventory(directory / "model.mlpackage") != manifest["package_files"]:
raise ValueError("Core ML package changed.")
return manifest
def convert(bundle, output, target):
ct = coremltools()
if output.exists():
raise ValueError("Output exists; choose a new conversion experiment.")
lm = BundleLM(bundle)
traced = trace_graph(lm.model)
model = ct.convert(traced, source="pytorch", convert_to="mlprogram",
minimum_deployment_target=getattr(ct.target, f"iOS{target}"),
compute_precision=ct.precision.FLOAT16,
inputs=[ct.TensorType(name="input_ids", shape=(1, ct.RangeDim(lower_bound=1,
upper_bound=lm.model.config.context_length, default=min(16, lm.model.config.context_length))), dtype=np.int32)],
outputs=[ct.TensorType(name="logits", dtype=np.float32)], skip_model_load=True)
model.short_description = "Frozen Tiny Japanese GPT; LM-only reranking and phrase continuation. No KV cache."
model.user_defined_metadata["bundle_manifest_sha256"] = lm.metadata["bundle_manifest_sha256"]
finish(output, model, {"kind": "fp16", "minimum_ios": target, "bundle": lm.metadata,
"interface": {"input_ids": "int32 [1,T], 1<=T<=128; right PAD only",
"logits": "float32 [1,T,16384]; FP16 computation; no softmax"}})
def compress(source, output, method, bits, group_size, block_size, validation):
ct = coremltools()
if output.exists():
raise ValueError("Output exists; choose a new compression experiment.")
original = verify_package(source)
if original["kind"] != "fp16":
raise ValueError("Compress the uncompressed FP16 source, not another compressed package.")
gate = read_json(validation)
if (gate.get("format") != "vimeml_alignment_v1" or not gate.get("passed") or
gate.get("coreml_manifest_sha256") != file_sha(source / "manifest.json")):
raise ValueError("First pass alignment for this exact FP16 model and supply its alignment.json.")
from coremltools.optimize import coreml as opt
model = ct.models.MLModel(str(source / "model.mlpackage"), skip_model_load=True)
if method == "palette":
grouped = group_size > 0
if grouped and original["minimum_ios"] < 18:
raise ValueError("Grouped-channel palettization requires an iOS18 source conversion.")
config = opt.OpPalettizerConfig(mode="kmeans", nbits=bits, weight_threshold=2048,
granularity="per_grouped_channel" if grouped else "per_tensor", group_size=group_size or 32)
result = opt.palettize_weights(model, config=opt.OptimizationConfig(global_config=config))
else:
if original["minimum_ios"] < 18:
raise ValueError("This blockwise linear compression experiment requires an iOS18 source.")
config = opt.OpLinearQuantizerConfig(mode="linear_symmetric", dtype=f"int{bits}",
granularity="per_block", block_size=block_size, weight_threshold=2048)
result = opt.linear_quantize_weights(model, config=opt.OptimizationConfig(global_config=config))
ops = inspect_spec(result)["operations"]
if not any(name.startswith("constexpr_") for name in ops):
raise ValueError("No compressed constexpr weights found; refusing to label this a compressed model.")
finish(output, result, {"kind": f"{method}{bits}", "minimum_ios": original["minimum_ios"],
"bundle": original["bundle"], "interface": original["interface"],
"source_manifest_sha256": file_sha(source / "manifest.json"),
"fp16_alignment_sha256": file_sha(validation),
"compression": {"method": method, "bits": bits, "group_size": group_size,
"block_size": block_size, "weight_threshold": 2048,
"scope": "Eligible constants; small constants remain uncompressed; no activation quantization or retraining."}})
class CoreMLForward:
def __init__(self, directory, config, compute_units):
ct = coremltools()
self.config = config
self.calls = 0
self.predict_seconds = 0.0
self.compute_units = compute_units
started = time.perf_counter()
self.model = ct.models.MLModel(str(Path(directory) / "model.mlpackage"),
compute_units=getattr(ct.ComputeUnit, compute_units))
self.load_seconds = time.perf_counter() - started
def __call__(self, inputs):
ids = inputs.detach().cpu().numpy()
if ids.ndim != 2 or ids.shape[0] < 1 or not 1 <= ids.shape[1] <= self.config.context_length:
raise ValueError("Expected nonempty batch with 1..128 tokens.")
if np.any(ids < 0) or np.any(ids >= self.config.vocab_size):
raise ValueError("Token ID outside vocabulary.")
outputs = []
# First deployment contract uses batch=1; preserve batched scoring semantics on host.
for row in ids:
started = time.perf_counter()
logits = np.array(self.model.predict({"input_ids": row[None].astype(np.int32)})["logits"],
dtype=np.float32, copy=True)
self.predict_seconds += time.perf_counter() - started
self.calls += 1
if logits.shape != (1, len(row), self.config.vocab_size) or not np.isfinite(logits).all():
raise ValueError("Core ML returned wrong shape or nonfinite logits.")
outputs.append(torch.from_numpy(logits))
return torch.cat(outputs)
class CoreMLLM(BundleLM):
def __init__(self, bundle, directory, compute_units="CPU_ONLY"):
super().__init__(bundle)
manifest = verify_package(directory)
if manifest["bundle"]["bundle_manifest_sha256"] != self.metadata["bundle_manifest_sha256"]:
raise ValueError("Core ML package belongs to a different inference bundle.")
config = self.model.config
self.model = CoreMLForward(directory, config, compute_units)
self.metadata.update(precision=manifest["kind"], compute_units=compute_units,
coreml_manifest_sha256=file_sha(Path(directory) / "manifest.json"))
|