{
"schema_version": 1,
"target": "MLX-5bit",
"source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
"validated_at": "2026-08-15T17:11:08.074526+00:00",
"structural": {
"passed": true,
"passes": [
"artifact directory exists",
"atomic build completion record",
"local SHA-256 manifest",
"all build-hashed files present (24)",
"all build payload SHA-256 hashes match",
"per-artifact quantization manifest",
"artifact manifest base model",
"artifact manifest source revision",
"artifact manifest declares vanilla quantization",
"plausible size 20.31 GB in [19, 24]",
"required sidecar config.json",
"required sidecar tokenizer_config.json",
"required sidecar generation_config.json",
"required sidecar preprocessor_config.json",
"required sidecar video_preprocessor_config.json",
"required sidecar chat_template.jinja",
"required sidecar tokenizer.json",
"required sidecar vocab.json",
"required sidecar merges.txt",
"required sidecar LICENSE",
"chat template byte-identical to source",
"generation_config.json semantically intact",
"preprocessor_config.json semantically intact",
"video_preprocessor_config.json semantically intact",
"tokenizer.json byte-identical to source",
"vocab.json byte-identical to source",
"merges.txt byte-identical to source",
"LICENSE byte-identical to source",
"tokenizer config preserves chat_template",
"tokenizer config preserves eos_token",
"tokenizer config preserves pad_token",
"tokenizer config preserves additional_special_tokens",
"official internal architecture id retained",
"MTP layer declaration retained",
"vision configuration retained",
"image special token id retained",
"video special token id retained",
"vision-start token id retained",
"vision-end token id retained",
"text architecture preserves num_hidden_layers",
"text architecture preserves hidden_size",
"text architecture preserves vocab_size",
"text architecture preserves max_position_embeddings",
"text architecture preserves full_attention_interval",
"text architecture preserves layer_types",
"text architecture preserves linear_conv_kernel_dim",
"text architecture preserves linear_key_head_dim",
"text architecture preserves linear_num_key_heads",
"text architecture preserves linear_num_value_heads",
"text architecture preserves linear_value_head_dim",
"full vision configuration retained",
"main safetensors readable",
"vision tensors retained (333)",
"main MLX bundle does not misplace MTP tensors",
"MTP drafter tensors retained (15)",
"MTP drafter model type",
"all floating tensors finite"
],
"failures": [],
"warnings": [],
"disk_size_bytes": 20308344457
},
"runtime": {
"runtime": "mlx-vlm",
"model": "/Users/oz/Documents/ChatGPT/qwen3.8-27b/qwen38-27b-quants/quantization-suite/artifacts/MLX-5bit",
"artifact_role": "quantized_candidate",
"source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
"text_tests": [
{
"id": "fact",
"category": "factual",
"passed": true,
"failures": [],
"text": "The capital of France is Paris.",
"finish_reason": "stop",
"prompt_tokens": 24,
"generation_tokens": 8,
"generation_tps": 11.631935751010623,
"peak_memory_gb": 21.51083095,
"wall_seconds": 1.797560500097461,
"loop_score": 0.0
},
{
"id": "math",
"category": "mathematical_reasoning",
"passed": true,
"failures": [],
"text": "The user is asking a simple multiplication problem: 6 rows of 7 blue marbles. I need to show a short calculation.\n\n6 × 7 = 42\n\n\n6 × 7 = **42** blue marbles.",
"finish_reason": "stop",
"prompt_tokens": 36,
"generation_tokens": 53,
"generation_tps": 10.64213086594602,
"peak_memory_gb": 21.601646682,
"wall_seconds": 5.313823957927525,
"loop_score": 0.0
},
{
"id": "code",
"category": "coding",
"passed": true,
"failures": [],
"text": "The user wants a Python function called `clamp(x, low, high)` that clamps a value `x` to be within the inclusive range `[low, high]`. This is a straightforward function:\n- If x < low, return low\n- If x > high, return high\n- Otherwise, return x\n\nI can implement this using `max(low, min(x, high))` or with if/elif/else. The one-liner with max/min is clean and Pythonic. Let me provide a single fenced code block.\n\n\n```python\ndef clamp(x, low, high):\n \"\"\"Return x limited to the inclusive range [low, high].\"\"\"\n return max(low, min(x, high))\n```",
"finish_reason": "stop",
"prompt_tokens": 67,
"generation_tokens": 155,
"generation_tps": 10.653914015405283,
"peak_memory_gb": 21.689669278,
"wall_seconds": 15.041777167003602,
"loop_score": 0.0
},
{
"id": "multiturn",
"category": "multi_turn",
"passed": true,
"failures": [],
"text": "cedar",
"finish_reason": "stop",
"prompt_tokens": 50,
"generation_tokens": 3,
"generation_tps": 15.956279792344992,
"peak_memory_gb": 21.689669278,
"wall_seconds": 0.5222798750037327,
"loop_score": 0.0
},
{
"id": "thinking_off",
"category": "thinking_disabled",
"passed": true,
"failures": [],
"text": "Yes, 17 is prime.\n\n**Reason:** It has no divisors other than 1 and itself (it is not divisible by 2, 3, or any other integer less than its square root, $\\sqrt{17} \\approx 4.12$).",
"finish_reason": "stop",
"prompt_tokens": 25,
"generation_tokens": 61,
"generation_tps": 10.82475589727582,
"peak_memory_gb": 21.689669278,
"wall_seconds": 5.99831187503878,
"loop_score": 0.0
},
{
"id": "reasoning_low",
"category": "reasoning_effort_low",
"passed": true,
"failures": [],
"text": "Compare 3/5 and 5/9.\n\nCross multiply: 3 × 9 = 27, 5 × 5 = 25.\n\nSince 27 > 25, 3/5 > 5/9.\n\n\n3/5",
"finish_reason": "stop",
"prompt_tokens": 60,
"generation_tokens": 62,
"generation_tps": 10.787043967920432,
"peak_memory_gb": 21.689669278,
"wall_seconds": 6.09620512498077,
"loop_score": 0.0
},
{
"id": "reasoning_high",
"category": "reasoning_effort_high",
"passed": true,
"failures": [],
"text": "We need answer user: Find smallest positive integer divisible by 6,8,15. Explain briefly. Need compute LCM. 6=2*3, 8=2^3, 15=3*5. LCM max powers: 2^3*3*5=120. Need final concise.\n\n\nThe smallest positive integer divisible by 6, 8, and 15 is their least common multiple (LCM).\n\nPrime factorizations:\n- 6 = 2 × 3\n- 8 = 2³\n- 15 = 3 × 5\n\nTake the highest power of each prime:\n- 2³, 3, 5\n\nLCM = 2³ × 3 × 5 = 8 × 3 × 5 = **120**.",
"finish_reason": "stop",
"prompt_tokens": 73,
"generation_tokens": 182,
"generation_tps": 10.689433777326002,
"peak_memory_gb": 21.708525014,
"wall_seconds": 17.524736167048104,
"loop_score": 0.0
},
{
"id": "long_form",
"category": "long_form",
"passed": true,
"failures": [],
"text": "Checksums serve as the foundational integrity layer for reproducible machine learning by providing cryptographic fingerprints for every component of the release, including raw datasets, preprocessed features, and model weights. By verifying that the hash of a downloaded artifact matches the published value, engineers can guarantee that the exact binary files used during training are identical to those deployed in production, eliminating subtle data drift or corruption that might otherwise introduce non-deterministic behavior into the inference pipeline.\n\nImmutable revisions extend this integrity to the code and configuration level by locking specific versions of source code, dependency libraries, and hyperparameters into a permanent, unchangeable state. Unlike mutable branches that can be updated or deleted, an immutable revision ensures that the precise combination of algorithms, framework versions, and environment settings used to generate a model is preserved forever, allowing any team member to reconstruct the exact experimental conditions without ambiguity or reliance on external, shifting dependencies.\n\nTogether, these mechanisms create a verifiable audit trail that links the final model artifact back to its specific inputs and processing logic. When a checksum validates the data and an immutable revision validates the code, the entire release becomes a self-contained, deterministic unit; this means that re-running the pipeline with the same inputs will always yield the same outputs, which is critical for debugging, regulatory compliance, and ensuring that performance metrics reported in development accurately reflect real-world behavior.\n\nUltimately, the synergy between checksums and immutable revisions transforms machine learning releases from fragile, environment-dependent scripts into robust, software-engineering-grade products. This approach reduces the \"it works on my machine\" problem by standardizing the definition of a release, thereby accelerating deployment cycles, simplifying rollback procedures, and fostering trust among stakeholders who rely on the consistency and transparency of the AI systems being delivered.",
"finish_reason": "stop",
"prompt_tokens": 31,
"generation_tokens": 353,
"generation_tps": 10.535428115142368,
"peak_memory_gb": 21.708525014,
"wall_seconds": 33.87512920808513,
"loop_score": 0.0
},
{
"id": "non_english",
"category": "non_english",
"passed": true,
"failures": [],
"text": "太陽光が大気中を通過する際、波長の短い青い光が他の色よりも強く散乱されるためです。\nこの散乱された青い光が私たちの目に入ってくることで、空は青く見えます。",
"finish_reason": "stop",
"prompt_tokens": 28,
"generation_tokens": 51,
"generation_tps": 10.657100073582797,
"peak_memory_gb": 21.708525014,
"wall_seconds": 5.154522541095503,
"loop_score": 0.0
},
{
"id": "json",
"category": "structured_json",
"passed": true,
"failures": [],
"text": "```json\n{\n \"name\": \"sample\",\n \"primes\": [2, 3, 5],\n \"active\": true\n}\n```",
"finish_reason": "stop",
"prompt_tokens": 41,
"generation_tokens": 38,
"generation_tps": 10.725316081928828,
"peak_memory_gb": 21.708525014,
"wall_seconds": 3.8777878750115633,
"loop_score": 0.0
}
],
"tool_tests": [
{
"id": "one_tool",
"passed": true,
"failures": [],
"calls": [
{
"name": "weather_at",
"arguments": {
"location": "Paris",
"units": "celsius"
}
}
],
"raw": "The user is asking to call weather_at for Paris in Celsius units.\n\n\n\n\n\nParis\n\n\ncelsius\n\n\n",
"finish_reason": "stop",
"dependent_raw": null
},
{
"id": "parallel_tools",
"passed": true,
"failures": [],
"calls": [
{
"name": "weather_at",
"arguments": {
"location": "Paris",
"units": "celsius"
}
},
{
"name": "weather_at",
"arguments": {
"location": "Austin",
"units": "fahrenheit"
}
}
],
"raw": "The user is asking to call weather_at in parallel for Paris (Celsius) and Austin (Fahrenheit). Since these are independent calls, they can be executed in the same block.\n\n\n\n\n\nParis\n\n\ncelsius\n\n\n\n\n\n\nAustin\n\n\nfahrenheit\n\n\n",
"finish_reason": "stop",
"dependent_raw": null
},
{
"id": "sequential_dependency",
"passed": true,
"failures": [],
"calls": [
{
"name": "weather_at",
"arguments": {
"location": "Paris",
"units": "celsius"
}
}
],
"raw": "The user is asking me to do the following:\n1. First, call weather_at for Paris in Celsius\n2. After receiving the result, create an event with a title that exactly matches the returned condition string\n\nSince the second step depends on the result of the first step, I need to call weather_at first and wait for the result before calling add_event.\n\n\n\n\n\nParis\n\n\ncelsius\n\n\n",
"finish_reason": "stop",
"dependent_raw": "The weather condition is \"sunny\". Next, I need to create an event with the title exactly \"sunny\".\n\n\n\n\n\nsunny\n\n\n"
},
{
"id": "nested_json",
"passed": true,
"failures": [],
"calls": [
{
"name": "add_event",
"arguments": {
"title": "Release review",
"attendees": [
"ana@example.test",
"li@example.test"
],
"metadata": {
"release": {
"model": "Qwen3.8-27B",
"slot": 21
}
}
}
}
],
"raw": "The user is asking to call add_event with specific parameters:\n- title: \"Release review\"\n- attendees: [\"ana@example.test\", \"li@example.test\"]\n- metadata: {\"release\":{\"model\":\"Qwen3.8-27B\",\"slot\":21}}\n\nThis is a straightforward request. Let's make the function call.\n\n\n\n\n\nRelease review\n\n\n[\"ana@example.test\", \"li@example.test\"]\n\n\n{\"release\": {\"model\": \"Qwen3.8-27B\", \"slot\": 21}}\n\n\n",
"finish_reason": "stop",
"dependent_raw": null
},
{
"id": "invalid_recovery",
"passed": true,
"failures": [],
"calls": [
{
"name": "weather_at",
"arguments": {
"location": "Paris",
"units": "celsius"
}
}
],
"raw": "The user is asking to call weather_at for Paris with units=kelvin. However, the schema only allows \"celsius\" or \"fahrenheit\" for units. Therefore, kelvin violates the schema. The user has explicitly stated: \"If that violates the schema, recover by using celsius.\"\n\nSo I should call weather_at with location=Paris, units=celsius.\n\nShould I first try kelvin? The instructions say to call with kelvin, and if it violates the schema, recover with celsius. Since I know the schema, I can directly call with celsius. But the user's instruction is to call with kelvin first. However, calling with an invalid enum value would be an error. The user has anticipated this and instructed to recover with celsius. The cleanest approach is: since kelvin is not in the enum, directly call with celsius.\n\nActually, let me think about whether I should attempt the kelvin call. The function schema has an enum constraint. If I call with kelvin, it will fail validation. The user has explicitly said to recover with celsius if it violates the schema. Since I can determine that it violates the schema without calling, I can just call with celsius directly. That's the sensible recovery.\n\nLet me make the call: weather_at(location=\"Paris\", units=\"celsius\").\n\n\n\n\n\nParis\n\n\ncelsius\n\n\n",
"finish_reason": "stop",
"dependent_raw": null
}
],
"vision_tests": [
{
"id": "shapes_colors",
"passed": true,
"missing_patterns": [],
"text": "Based on the image provided, here are the three large shapes from left to right with their corresponding colors:\n\n1. **Square** — **Red**\n2. **Circle** — **Blue**\n3. **Triangle** — **Green**\n\n*(Note: The bar chart below shows different colors for bars A, B, and C — orange, purple, and teal — but these are not the “large shapes” referenced in your question. The large shapes are clearly labeled with their color names beneath them.)*\n\n✅ Final Answer:\n- **Red Square**\n- **Blue Circle**\n- **Green Triangle**",
"finish_reason": "stop"
},
{
"id": "printed_text",
"passed": true,
"missing_patterns": [],
"text": "VISION CHECK 27B",
"finish_reason": "stop"
},
{
"id": "chart",
"passed": true,
"missing_patterns": [],
"text": "Based on the provided image:\n\n- The bar chart has three bars labeled **A**, **B**, and **C**.\n- The numbers printed above each bar are:\n - **A**: 60\n - **B**: 105\n - **C**: 135\n\nThe tallest bar is **C**, as it reaches the highest point on the y-axis, and the number printed above it is **135**.\n\n✅ **Answer: Bar C is the tallest, and the number printed above it is 135.**",
"finish_reason": "stop"
}
],
"mtp": {
"passed": true,
"drafter_kind": "mtp",
"output_equivalent_temperature_zero": true,
"accepted_drafts": 84,
"drafted_tokens": 88,
"acceptance_rate": 0.9545454545454546,
"baseline_tps": 10.722035256649095,
"mtp_tps": 13.000213911486806,
"speedup": 1.2124763256514137,
"measured_improvement": true,
"baseline_wall_seconds": 12.302287542028353,
"mtp_wall_seconds": 10.11357758298982,
"advertise_acceleration": true
},
"warnings": [],
"phases": [
"mtp",
"text",
"tools",
"vision"
],
"validation_inputs": {
"prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528",
"tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122",
"image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295"
},
"metal_memory_policy": {
"device": {
"device_name": "Apple M5 Pro",
"max_recommended_working_set_size": 55662788608,
"memory_size": 68719476736,
"architecture": "applegpu_g17s",
"max_buffer_length": 41747087360,
"resource_limit": 499000
},
"cache_limit_bytes": 256000000,
"wired_limit_bytes": 54549532835,
"previous_cache_limit_bytes": 65283502899,
"previous_wired_limit_bytes": 0,
"warnings": []
}
},
"warnings": [],
"overall_passed": true,
"runtime_failures": [],
"quality": {
"schema_version": 1,
"comparison_type": "cross-runtime output agreement against pinned BF16 source",
"passed": true,
"source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
"embedding_model": {
"repo_id": "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
"revision": "e8f8c211226b894fcb81acc59f3b34ba3efd5f42",
"pooling": "attention-mask mean pooling followed by L2 normalization",
"maximum_tokens": 256
},
"thresholds": {
"mean_semantic_similarity": 0.55,
"per_case_severe_regression": 0.25
},
"baseline_valid": true,
"candidate_functional": true,
"semantic_gate_passed": true,
"validation_inputs_match": true,
"validation_inputs": {
"prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528",
"tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122",
"image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295"
},
"mean_semantic_similarity": 0.9520436644554138,
"exact_matches": 4,
"comparisons": [
{
"id": "fact",
"reference_passed": true,
"candidate_passed": true,
"exact_match": true,
"sequence_agreement": 1.0,
"semantic_similarity": 1.0,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "math",
"reference_passed": true,
"candidate_passed": true,
"exact_match": false,
"sequence_agreement": 0.978328173374613,
"semantic_similarity": 0.9986979961395264,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "code",
"reference_passed": true,
"candidate_passed": true,
"exact_match": false,
"sequence_agreement": 0.9264565425023877,
"semantic_similarity": 0.9722706079483032,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "multiturn",
"reference_passed": true,
"candidate_passed": true,
"exact_match": true,
"sequence_agreement": 1.0,
"semantic_similarity": 1.0,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "thinking_off",
"reference_passed": true,
"candidate_passed": true,
"exact_match": false,
"sequence_agreement": 0.9695290858725761,
"semantic_similarity": 0.9862716794013977,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "reasoning_low",
"reference_passed": true,
"candidate_passed": true,
"exact_match": false,
"sequence_agreement": 0.6153846153846154,
"semantic_similarity": 0.807137131690979,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "reasoning_high",
"reference_passed": true,
"candidate_passed": true,
"exact_match": true,
"sequence_agreement": 1.0,
"semantic_similarity": 1.0,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "long_form",
"reference_passed": true,
"candidate_passed": true,
"exact_match": false,
"sequence_agreement": 0.23614973262032085,
"semantic_similarity": 0.8014230132102966,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "non_english",
"reference_passed": true,
"candidate_passed": true,
"exact_match": false,
"sequence_agreement": 0.6808510638297872,
"semantic_similarity": 0.9546362161636353,
"severe_regression": false,
"candidate_loop_score": 0.0
},
{
"id": "json",
"reference_passed": true,
"candidate_passed": true,
"exact_match": true,
"sequence_agreement": 1.0,
"semantic_similarity": 1.0,
"severe_regression": false,
"candidate_loop_score": 0.0
}
],
"functional_results": {
"reference_text": {
"passed": 10,
"total": 10
},
"candidate_text": {
"passed": 10,
"total": 10
},
"reference_tools": {
"passed": 5,
"total": 5
},
"candidate_tools": {
"passed": 5,
"total": 5
},
"reference_vision": {
"passed": 3,
"total": 3
},
"candidate_vision": {
"passed": 3,
"total": 3
}
},
"measurements": {
"average_generation_tps": 11.310333833788317,
"peak_memory_gb": 21.708525014,
"artifact_bytes": 20308322354,
"maximum_prompt_tokens_tested": 73,
"loop_rate": 0.0
},
"warnings": [
"Semantic similarity is a measured embedding-model proxy, not ground-truth accuracy.",
"Raw-logit equality is unavailable across all target runtimes; exact functional gates and output agreement are used for portable release validation.",
"Sequence agreement is lexical and is reported diagnostically, not used as semantic accuracy."
]
},
"quality_failures": []
}