{ "schema_version": 1, "target": "MLX-oQ5", "source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "validated_at": "2026-08-16T01:56:52.963372+00:00", "structural": { "passed": true, "passes": [ "artifact directory exists", "atomic build completion record", "local SHA-256 manifest", "all build-hashed files present (18)", "all build payload SHA-256 hashes match", "per-artifact quantization manifest", "artifact manifest base model", "artifact manifest source revision", "artifact manifest declares vanilla quantization", "plausible size 20.31 GB in [19, 28]", "required sidecar config.json", "required sidecar tokenizer_config.json", "required sidecar generation_config.json", "required sidecar preprocessor_config.json", "required sidecar video_preprocessor_config.json", "required sidecar chat_template.jinja", "required sidecar tokenizer.json", "required sidecar vocab.json", "required sidecar merges.txt", "required sidecar LICENSE", "chat template byte-identical to source", "generation_config.json semantically intact", "preprocessor_config.json semantically intact", "video_preprocessor_config.json semantically intact", "tokenizer.json byte-identical to source", "vocab.json byte-identical to source", "merges.txt byte-identical to source", "LICENSE byte-identical to source", "tokenizer config preserves chat_template", "tokenizer config preserves eos_token", "tokenizer config preserves pad_token", "tokenizer config preserves additional_special_tokens", "official internal architecture id retained", "MTP layer declaration retained", "vision configuration retained", "image special token id retained", "video special token id retained", "vision-start token id retained", "vision-end token id retained", "text architecture preserves num_hidden_layers", "text architecture preserves hidden_size", "text architecture preserves vocab_size", "text architecture preserves max_position_embeddings", "text architecture preserves full_attention_interval", "text architecture preserves layer_types", "text architecture preserves linear_conv_kernel_dim", "text architecture preserves linear_key_head_dim", "text architecture preserves linear_num_key_heads", "text architecture preserves linear_num_value_heads", "text architecture preserves linear_value_head_dim", "full vision configuration retained", "main safetensors readable", "vision tensors retained (333)", "native MTP tensors retained in oMLX bundle (29)", "all floating tensors finite" ], "failures": [], "warnings": [], "disk_size_bytes": 20314063540 }, "runtime": { "runtime": "oMLX", "model": "/Users/oz/Documents/ChatGPT/qwen3.8-27b/qwen38-27b-quants/quantization-suite/artifacts/MLX-oQ5", "text_tests": [ { "id": "fact", "category": "factual", "passed": true, "failures": [], "text": "The capital of France is Paris.", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 24, "completion_tokens": 7, "total_tokens": 31, "input_tokens": 24, "output_tokens": 7, "prompt_tokens_details": { "cached_tokens": 0 }, "model_load_duration": 2.24, "total_time": 1.07 }, "timings": null, "wall_seconds": 3.3369667909573764, "loop_score": 0.0 }, { "id": "math", "category": "mathematical_reasoning", "passed": true, "failures": [], "text": "**6 × 7 = 42**\n\nThere are **42** blue marbles.", "reasoning_content": "The user is asking a simple multiplication problem: 6 rows of 7 blue marbles. I need to show a short calculation.\n\n6 × 7 = 42", "finish_reason": "stop", "usage": { "prompt_tokens": 36, "completion_tokens": 60, "total_tokens": 96, "input_tokens": 36, "output_tokens": 60, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 2.6 }, "timings": null, "wall_seconds": 2.6100297500379384, "loop_score": 0.0 }, { "id": "code", "category": "coding", "passed": true, "failures": [], "text": "```python\ndef clamp(x, low, high):\n return max(low, min(x, high))\n```", "reasoning_content": "The user wants a Python function called `clamp(x, low, high)` that clamps a value `x` to be within the inclusive range `[low, high]`. This is a straightforward function:\n- If x < low, return low\n- If x > high, return high\n- Otherwise, return x\n\nI can implement this using `max(low, min(x, high))` or with if/elif/else. The one-liner with max/min is clean and Pythonic.\n\nThey want one fenced code block.", "finish_reason": "stop", "usage": { "prompt_tokens": 67, "completion_tokens": 137, "total_tokens": 204, "input_tokens": 67, "output_tokens": 137, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 5.79 }, "timings": null, "wall_seconds": 5.801307375077158, "loop_score": 0.0 }, { "id": "multiturn", "category": "multi_turn", "passed": true, "failures": [], "text": "cedar", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 50, "completion_tokens": 2, "total_tokens": 52, "input_tokens": 50, "output_tokens": 2, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 0.61 }, "timings": null, "wall_seconds": 0.6155566251836717, "loop_score": 0.0 }, { "id": "thinking_off", "category": "thinking_disabled", "passed": true, "failures": [], "text": "Yes, 17 is prime.\n\n**Reason:** It has no divisors other than 1 and itself (it is not divisible by 2, 3, or any other integer less than its square root, $\\sqrt{17} \\approx 4.12$).", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 25, "completion_tokens": 60, "total_tokens": 85, "input_tokens": 25, "output_tokens": 60, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 2.58 }, "timings": null, "wall_seconds": 2.5851613751146942, "loop_score": 0.0 }, { "id": "reasoning_low", "category": "reasoning_effort_low", "passed": true, "failures": [], "text": "3/5", "reasoning_content": "Compare 3/5 and 5/9.\n\nCross multiply: 3 × 9 = 27, 5 × 5 = 25.\n\nSince 27 > 25, 3/5 > 5/9.", "finish_reason": "stop", "usage": { "prompt_tokens": 60, "completion_tokens": 61, "total_tokens": 121, "input_tokens": 60, "output_tokens": 61, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 2.45 }, "timings": null, "wall_seconds": 2.4517624999862164, "loop_score": 0.0 }, { "id": "reasoning_high", "category": "reasoning_effort_high", "passed": true, "failures": [], "text": "The smallest positive integer divisible by 6, 8, and 15 is their least common multiple (LCM).\n\nPrime factorizations:\n- 6 = 2 × 3\n- 8 = 2³\n- 15 = 3 × 5\n\nTake the highest power of each prime:\n- 2³, 3, 5\n\nLCM = 2³ × 3 × 5 = 8 × 3 × 5 = **120**.", "reasoning_content": "We need answer user: Find smallest positive integer divisible by 6,8,15. Explain briefly. Need compute LCM. 6=2*3, 8=2^3, 15=3*5. LCM max powers: 2^3*3*5=120. Need final concise.", "finish_reason": "stop", "usage": { "prompt_tokens": 73, "completion_tokens": 181, "total_tokens": 254, "input_tokens": 73, "output_tokens": 181, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 6.34 }, "timings": null, "wall_seconds": 6.350575708085671, "loop_score": 0.0 }, { "id": "long_form", "category": "long_form", "passed": true, "failures": [], "text": "Checksums serve as the foundational integrity layer for reproducible machine learning by providing cryptographic fingerprints for every component of the release, including raw datasets, preprocessed features, and model weights. By verifying that the hash of a downloaded artifact matches the published value, engineers can guarantee that the exact data and code used to train a model have not been altered, corrupted, or tampered with during transfer. This deterministic verification eliminates ambiguity regarding the input state, ensuring that any team member or automated pipeline can confirm they are working with the precise version of the data required to replicate the original training environment.\n\nImmutable revisions extend this integrity to the code and configuration layers by locking specific versions of source code, dependencies, and hyperparameters into unchangeable snapshots. Unlike mutable branches that can be updated or overwritten, an immutable revision ensures that the exact logic, library versions, and training parameters used to generate a model are preserved forever. This prevents the \"works on my machine\" problem by decoupling the release from the evolving state of development repositories, allowing researchers to trace back any specific model artifact to the exact commit and environment configuration that produced it.\n\nTogether, these mechanisms create a verifiable chain of custody that enables true end-to-end reproducibility. When a checksum-verified dataset is processed by an immutable code revision, the resulting model artifact becomes a deterministic function of its inputs, meaning that re-running the pipeline with the same verified components will yield identical results. This transparency is critical for debugging, as it allows teams to isolate whether a performance discrepancy stems from data drift, code changes, or environmental differences, rather than facing an opaque black box where multiple variables may have shifted simultaneously.\n\nUltimately, integrating checksums and immutable revisions transforms ML releases from fragile, ad-hoc experiments into robust, auditable software products. This approach satisfies regulatory and compliance requirements by providing a clear, tamper-evident history of how models were built, which is essential for high-stakes applications in finance, healthcare, and autonomous systems. By standardizing these practices, organizations reduce the risk of silent failures, accelerate onboarding for new team members, and build trust in the reliability and fairness of their machine learning systems.", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 31, "completion_tokens": 436, "total_tokens": 467, "input_tokens": 31, "output_tokens": 436, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 19.93 }, "timings": null, "wall_seconds": 19.933278792072088, "loop_score": 0.0 }, { "id": "non_english", "category": "non_english", "passed": true, "failures": [], "text": "太陽光が大気中を通過する際、波長の短い青い光が他の色よりも強く散乱されます。そのため、私たちの目には空全体が青く見えます。", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 28, "completion_tokens": 39, "total_tokens": 67, "input_tokens": 28, "output_tokens": 39, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 2.64 }, "timings": null, "wall_seconds": 2.6417087910231203, "loop_score": 0.0 }, { "id": "json", "category": "structured_json", "passed": true, "failures": [], "text": "{\n \"name\": \"sample\",\n \"primes\": [2, 3, 5],\n \"active\": true\n}", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 41, "completion_tokens": 32, "total_tokens": 73, "input_tokens": 41, "output_tokens": 32, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 1.58 }, "timings": null, "wall_seconds": 1.5874647919554263, "loop_score": 0.0 } ], "tool_tests": [ { "id": "one_tool", "passed": true, "failures": [], "calls": [ { "id": "call_f6033b1f", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null }, { "id": "parallel_tools", "passed": true, "failures": [], "calls": [ { "id": "call_3661aafa", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } }, { "id": "call_748b84f2", "name": "weather_at", "arguments": { "location": "Austin", "units": "fahrenheit" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null }, { "id": "sequential_dependency", "passed": true, "failures": [], "calls": [ { "id": "call_8b17646d", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": "" }, { "id": "nested_json", "passed": true, "failures": [], "calls": [ { "id": "call_30818c89", "name": "add_event", "arguments": { "title": "Release review", "attendees": [ "ana@example.test", "li@example.test" ], "metadata": { "release": { "model": "Qwen3.8-27B", "slot": 21 } } } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null }, { "id": "invalid_recovery", "passed": true, "failures": [], "calls": [ { "id": "call_2df8dbed", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null } ], "vision_tests": [ { "id": "shapes_colors", "passed": true, "missing_patterns": [], "text": "Based on the image provided, here are the three large shapes from left to right with their corresponding colors:\n\n1. **Square** — **Red**\n2. **Circle** — **Blue**\n3. **Triangle** — **Green**\n\n*(Note: The bar chart below shows different colors for bars A, B, and C — orange, purple, and teal — but the question specifically asks for the “three large shapes” at the top, which are labeled RED, BLUE, and GREEN.)*\n\n✅ Final Answer:\n- Square (Red)\n- Circle (Blue)\n- Triangle (Green)", "finish_reason": "stop" }, { "id": "printed_text", "passed": true, "missing_patterns": [], "text": "VISION CHECK 27B", "finish_reason": "stop" }, { "id": "chart", "passed": true, "missing_patterns": [], "text": "Based on the provided image:\n\n- The bar chart has three bars labeled **A**, **B**, and **C**.\n- The numbers printed above each bar are:\n - **A**: 60\n - **B**: 105\n - **C**: 135\n\nThe tallest bar is **C**, as it reaches the highest point on the y-axis, and the number printed above it is **135**.\n\n✅ **Answer: Bar C is the tallest, and the number printed above it is 135.**", "finish_reason": "stop" } ], "mtp": { "passed": true, "backend": "oMLX Lightning MTP (integrated source MTP head)", "output_equivalent_temperature_zero": true, "baseline": { "text": "To list the first ten square numbers, we calculate the square of each integer from 1 to 10 ($n^2$).\n\n1. $1^2 = 1$\n2. $2^2 = 4$\n3. $3^2 = 9$\n4. $4^2 = 16$\n5. $5^2 = 25$\n6. $6^2 = 36$\n7. $7^2 = 49$\n8. $8^2 = 64", "finish_reason": "length", "usage": { "prompt_tokens": 21, "completion_tokens": 128, "total_tokens": 149, "input_tokens": 21, "output_tokens": 128, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 10.37 }, "wall_seconds": 10.375257540959865, "generation_tps": 12.343297974927676 }, "mtp_measurement": { "text": "To list the first ten square numbers, we calculate the square of each integer from 1 to 10 ($n^2$).\n\n1. $1^2 = 1$\n2. $2^2 = 4$\n3. $3^2 = 9$\n4. $4^2 = 16$\n5. $5^2 = 25$\n6. $6^2 = 36$\n7. $7^2 = 49$\n8. $8^2 = 64", "finish_reason": "length", "usage": { "prompt_tokens": 21, "completion_tokens": 128, "total_tokens": 149, "input_tokens": 21, "output_tokens": 128, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 4.23 }, "wall_seconds": 4.238896624883637, "generation_tps": 30.260047281323875 }, "baseline_tps": 12.343297974927676, "mtp_tps": 30.260047281323875, "speedup": 2.4515366430260044, "measured_improvement": true, "advertise_acceleration": true, "native_stats": { "finish_reason": "length", "tokens": 128, "cycles": 36, "tokens_per_cycle": 3.56, "accepted_drafts": 93, "drafted_tokens": 96, "acceptance_rate": 0.96875 }, "failure": null }, "peak_process_rss_gb": 21.0919424, "validation_inputs": { "prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528", "tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122", "image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295" }, "warnings": [ "Peak process RSS is a unified-memory process measurement, not a separate VRAM measurement." ] }, "warnings": [], "overall_passed": true, "runtime_failures": [], "quality": { "schema_version": 1, "comparison_type": "cross-runtime output agreement against pinned BF16 source", "passed": true, "source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "embedding_model": { "repo_id": "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2", "revision": "e8f8c211226b894fcb81acc59f3b34ba3efd5f42", "pooling": "attention-mask mean pooling followed by L2 normalization", "maximum_tokens": 256 }, "thresholds": { "mean_semantic_similarity": 0.55, "per_case_severe_regression": 0.25 }, "baseline_valid": true, "candidate_functional": true, "semantic_gate_passed": true, "validation_inputs_match": true, "validation_inputs": { "prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528", "tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122", "image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295" }, "mean_semantic_similarity": 0.8852966785430908, "exact_matches": 2, "comparisons": [ { "id": "fact", "reference_passed": true, "candidate_passed": true, "exact_match": true, "sequence_agreement": 1.0, "semantic_similarity": 1.0, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "math", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.33170731707317075, "semantic_similarity": 0.7768455147743225, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "code", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.25622775800711745, "semantic_similarity": 0.867202639579773, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "multiturn", "reference_passed": true, "candidate_passed": true, "exact_match": true, "sequence_agreement": 1.0, "semantic_similarity": 1.0, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "thinking_off", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.9695290858725761, "semantic_similarity": 0.9862716794013977, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "reasoning_low", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.0625, "semantic_similarity": 0.6260854005813599, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "reasoning_high", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.7183308494783904, "semantic_similarity": 0.8893308639526367, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "long_form", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.28044714649931357, "semantic_similarity": 0.8436300754547119, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "non_english", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.896, "semantic_similarity": 0.9634965658187866, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "json", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.9130434782608695, "semantic_similarity": 0.9001040458679199, "severe_regression": false, "candidate_loop_score": 0.0 } ], "functional_results": { "reference_text": { "passed": 10, "total": 10 }, "candidate_text": { "passed": 10, "total": 10 }, "reference_tools": { "passed": 5, "total": 5 }, "candidate_tools": { "passed": 5, "total": 5 }, "reference_vision": { "passed": 3, "total": 3 }, "candidate_vision": { "passed": 3, "total": 3 } }, "measurements": { "average_generation_tps": 19.016428184965864, "peak_memory_gb": 21.0919424, "artifact_bytes": 20314066494, "maximum_prompt_tokens_tested": 73, "loop_rate": 0.0 }, "warnings": [ "Semantic similarity is a measured embedding-model proxy, not ground-truth accuracy.", "Raw-logit equality is unavailable across all target runtimes; exact functional gates and output agreement are used for portable release validation.", "Sequence agreement is lexical and is reported diagnostically, not used as semantic accuracy." ] }, "quality_failures": [] }