File size: 6,534 Bytes
29f25be
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
"""Distribution-level check of a Core ML package against the FP32 bundle.

Distribution diagnostics supplement the strict logits, PAD and ranking checks.
They do not establish exact conversion equivalence. This reports KL(FP32 || model),
NLL on real text, top-1 agreement and, as a control, the same metrics for a
PyTorch simulation of symmetric INT8 block quantization. Read-only.
"""
import argparse
import copy
import json
import sys
from pathlib import Path

ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT / "src"))

import numpy as np
import torch

from vimeml.deployment.bundle import BundleLM, environment
from vimeml.deployment.coreml import verify_package
from vimeml.training.data import file_sha


def simulate_int8(model, block):
    """Mirror the conservative recipe: FP16 storage except QKV/positions, then INT8 per block."""
    if block <= 0 or any(p.ndim == 2 and p.shape[1] % block for p in model.parameters()):
        raise ValueError("Block size must be positive and divide every matrix input dimension.")
    result = copy.deepcopy(model)
    with torch.no_grad():
        for name, parameter in result.named_parameters():
            if parameter.ndim != 2:
                continue
            weight = parameter if ("qkv" in name or "position" in name) else parameter.half().float()
            rows, columns = weight.shape
            blocks = weight.reshape(rows, columns // block, block)
            scale = blocks.abs().amax(-1, keepdim=True).clamp_min(1e-12) / 127
            parameter.copy_((torch.round(blocks / scale).clamp(-127, 127) * scale).reshape(rows, columns))
    return result


def texts(benchmark):
    items = json.loads((benchmark / "evaluation_items.json").read_text(encoding="utf-8"))
    return [(item.get("context_text") or "") + item["expected_output"][0] for item in items]


def compare(lm, model, mlmodel, sequences):
    metrics = {key: [] for key in ("kl_coreml", "kl_simulated", "nll_fp32", "nll_coreml",
                                   "top1_agreement", "max_logprob_diff")}
    for ids in sequences:
        tensor = torch.tensor([ids])
        with torch.no_grad():
            reference = lm.model(tensor)[0].float().log_softmax(-1)
            simulated = model(tensor)[0].float().log_softmax(-1)
        logits = mlmodel.predict({"input_ids": np.array([ids], dtype=np.int32)})["logits"][0]
        coreml = torch.tensor(logits).log_softmax(-1)
        probabilities = reference.exp()
        metrics["kl_coreml"].append((probabilities * (reference - coreml)).sum(-1).mean().item())
        metrics["kl_simulated"].append((probabilities * (reference - simulated)).sum(-1).mean().item())
        targets = tensor[0, 1:, None]
        metrics["nll_fp32"].append(-reference[:-1].gather(-1, targets).mean().item())
        metrics["nll_coreml"].append(-coreml[:-1].gather(-1, targets).mean().item())
        metrics["top1_agreement"].append((reference.argmax(-1) == coreml.argmax(-1)).float().mean().item())
        metrics["max_logprob_diff"].append((reference - coreml).abs().max().item())
    return {key: {"mean": float(np.mean(values)), "max": float(np.max(values))} for key, values in metrics.items()}


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--bundle", type=Path, default=ROOT / "artifacts/deployment/tiny-ja-v1-inference-v1")
    parser.add_argument("--model", type=Path, required=True, help="Core ML experiment directory with model.mlpackage")
    parser.add_argument("--benchmark", type=Path, default=ROOT / "artifacts/benchmarks/ajimee-jwtd-v2-v1")
    parser.add_argument("--block-size", type=int, default=32)
    parser.add_argument("--compute-units", choices=("CPU_ONLY", "CPU_AND_GPU", "CPU_AND_NE", "ALL"), default="CPU_ONLY")
    parser.add_argument("--output", type=Path, help="Optional JSON report path; must not exist.")
    args = parser.parse_args()
    if args.output and args.output.exists():
        parser.error("Output exists.")
    lm = BundleLM(args.bundle)
    manifest = verify_package(args.model)
    if manifest["bundle"]["bundle_manifest_sha256"] != lm.metadata["bundle_manifest_sha256"]:
        raise ValueError("Core ML model and FP32 bundle identities differ.")
    if sys.platform != "darwin":
        parser.error("Run Core ML prediction manually on Mac.")
    import coremltools as ct
    torch.set_num_threads(4)
    mlmodel = ct.models.MLModel(str(args.model / "model.mlpackage"),
                                compute_units=getattr(ct.ComputeUnit, args.compute_units))
    simulated = simulate_int8(lm.model, args.block_size).eval()
    limit = lm.model.config.context_length - 1
    encoded = [lm.processor.encode(text) for text in texts(args.benchmark)]
    if not encoded or any(not ids for ids in encoded):
        raise ValueError("Benchmark must have at least one nonempty target token per example.")
    real = [[lm.special["bos"], *ids[:limit]] for ids in encoded]
    rng = np.random.default_rng(0)
    random = [[lm.special["bos"], *rng.integers(4, lm.model.config.vocab_size, limit).tolist()] for _ in range(5)]
    report = {"format": "vimeml_distribution_diagnostic_v2", "model": str(args.model),
              "compute_units": args.compute_units, "environment": environment(),
              "bundle_manifest_sha256": lm.metadata["bundle_manifest_sha256"],
              "coreml_manifest_sha256": file_sha(args.model / "manifest.json"),
              "benchmark_items_sha256": file_sha(args.benchmark / "evaluation_items.json"),
              "simulation": {"block_size": args.block_size,
                  "note": "FP32 scales and symmetric +/-127 rounding; not exact Core ML scale storage or kernels."},
              "sampling": {"examples": len(encoded), "content_token_limit": limit,
                  "truncated_examples": sum(len(ids) > limit for ids in encoded),
                  "aggregation": "Mean of each example's position mean; not token-weighted corpus mean.",
                  "kl_top1_include_last_position": True, "nll_excludes_last_position": True,
                  "random_sequences": 5, "random_seed": 0},
              "benchmark_text": compare(lm, simulated, mlmodel, real),
              "random_tokens": compare(lm, simulated, mlmodel, random)}
    text = json.dumps(report, ensure_ascii=False, indent=2)
    print(text)
    if args.output:
        args.output.parent.mkdir(parents=True, exist_ok=True)
        args.output.write_text(text + "\n", encoding="utf-8")


if __name__ == "__main__":
    main()