vimeml-tiny-ja-v2.1 / source /scripts /deployment /compare_quantization.py
Voltline's picture
Release VimeML V2.1 step40000 FP32 and Core ML INT8 (GPL-2.0)
29f25be verified
Raw History Blame Contribute Delete
6.53 kB
"""Distribution-level check of a Core ML package against the FP32 bundle.
Distribution diagnostics supplement the strict logits, PAD and ranking checks.
They do not establish exact conversion equivalence. This reports KL(FP32 || model),
NLL on real text, top-1 agreement and, as a control, the same metrics for a
PyTorch simulation of symmetric INT8 block quantization. Read-only.
"""
import argparse
import copy
import json
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT / "src"))
import numpy as np
import torch
from vimeml.deployment.bundle import BundleLM, environment
from vimeml.deployment.coreml import verify_package
from vimeml.training.data import file_sha
def simulate_int8(model, block):
"""Mirror the conservative recipe: FP16 storage except QKV/positions, then INT8 per block."""
if block <= 0 or any(p.ndim == 2 and p.shape[1] % block for p in model.parameters()):
raise ValueError("Block size must be positive and divide every matrix input dimension.")
result = copy.deepcopy(model)
with torch.no_grad():
for name, parameter in result.named_parameters():
if parameter.ndim != 2:
continue
weight = parameter if ("qkv" in name or "position" in name) else parameter.half().float()
rows, columns = weight.shape
blocks = weight.reshape(rows, columns // block, block)
scale = blocks.abs().amax(-1, keepdim=True).clamp_min(1e-12) / 127
parameter.copy_((torch.round(blocks / scale).clamp(-127, 127) * scale).reshape(rows, columns))
return result
def texts(benchmark):
items = json.loads((benchmark / "evaluation_items.json").read_text(encoding="utf-8"))
return [(item.get("context_text") or "") + item["expected_output"][0] for item in items]
def compare(lm, model, mlmodel, sequences):
metrics = {key: [] for key in ("kl_coreml", "kl_simulated", "nll_fp32", "nll_coreml",
"top1_agreement", "max_logprob_diff")}
for ids in sequences:
tensor = torch.tensor([ids])
with torch.no_grad():
reference = lm.model(tensor)[0].float().log_softmax(-1)
simulated = model(tensor)[0].float().log_softmax(-1)
logits = mlmodel.predict({"input_ids": np.array([ids], dtype=np.int32)})["logits"][0]
coreml = torch.tensor(logits).log_softmax(-1)
probabilities = reference.exp()
metrics["kl_coreml"].append((probabilities * (reference - coreml)).sum(-1).mean().item())
metrics["kl_simulated"].append((probabilities * (reference - simulated)).sum(-1).mean().item())
targets = tensor[0, 1:, None]
metrics["nll_fp32"].append(-reference[:-1].gather(-1, targets).mean().item())
metrics["nll_coreml"].append(-coreml[:-1].gather(-1, targets).mean().item())
metrics["top1_agreement"].append((reference.argmax(-1) == coreml.argmax(-1)).float().mean().item())
metrics["max_logprob_diff"].append((reference - coreml).abs().max().item())
return {key: {"mean": float(np.mean(values)), "max": float(np.max(values))} for key, values in metrics.items()}
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--bundle", type=Path, default=ROOT / "artifacts/deployment/tiny-ja-v1-inference-v1")
parser.add_argument("--model", type=Path, required=True, help="Core ML experiment directory with model.mlpackage")
parser.add_argument("--benchmark", type=Path, default=ROOT / "artifacts/benchmarks/ajimee-jwtd-v2-v1")
parser.add_argument("--block-size", type=int, default=32)
parser.add_argument("--compute-units", choices=("CPU_ONLY", "CPU_AND_GPU", "CPU_AND_NE", "ALL"), default="CPU_ONLY")
parser.add_argument("--output", type=Path, help="Optional JSON report path; must not exist.")
args = parser.parse_args()
if args.output and args.output.exists():
parser.error("Output exists.")
lm = BundleLM(args.bundle)
manifest = verify_package(args.model)
if manifest["bundle"]["bundle_manifest_sha256"] != lm.metadata["bundle_manifest_sha256"]:
raise ValueError("Core ML model and FP32 bundle identities differ.")
if sys.platform != "darwin":
parser.error("Run Core ML prediction manually on Mac.")
import coremltools as ct
torch.set_num_threads(4)
mlmodel = ct.models.MLModel(str(args.model / "model.mlpackage"),
compute_units=getattr(ct.ComputeUnit, args.compute_units))
simulated = simulate_int8(lm.model, args.block_size).eval()
limit = lm.model.config.context_length - 1
encoded = [lm.processor.encode(text) for text in texts(args.benchmark)]
if not encoded or any(not ids for ids in encoded):
raise ValueError("Benchmark must have at least one nonempty target token per example.")
real = [[lm.special["bos"], *ids[:limit]] for ids in encoded]
rng = np.random.default_rng(0)
random = [[lm.special["bos"], *rng.integers(4, lm.model.config.vocab_size, limit).tolist()] for _ in range(5)]
report = {"format": "vimeml_distribution_diagnostic_v2", "model": str(args.model),
"compute_units": args.compute_units, "environment": environment(),
"bundle_manifest_sha256": lm.metadata["bundle_manifest_sha256"],
"coreml_manifest_sha256": file_sha(args.model / "manifest.json"),
"benchmark_items_sha256": file_sha(args.benchmark / "evaluation_items.json"),
"simulation": {"block_size": args.block_size,
"note": "FP32 scales and symmetric +/-127 rounding; not exact Core ML scale storage or kernels."},
"sampling": {"examples": len(encoded), "content_token_limit": limit,
"truncated_examples": sum(len(ids) > limit for ids in encoded),
"aggregation": "Mean of each example's position mean; not token-weighted corpus mean.",
"kl_top1_include_last_position": True, "nll_excludes_last_position": True,
"random_sequences": 5, "random_seed": 0},
"benchmark_text": compare(lm, simulated, mlmodel, real),
"random_tokens": compare(lm, simulated, mlmodel, random)}
text = json.dumps(report, ensure_ascii=False, indent=2)
print(text)
if args.output:
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(text + "\n", encoding="utf-8")
if __name__ == "__main__":
main()