{ "schema_version": 1, "target": "MLX-oQ8", "source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "validated_at": "2026-08-16T02:02:31.543544+00:00", "structural": { "passed": true, "passes": [ "artifact directory exists", "atomic build completion record", "local SHA-256 manifest", "all build-hashed files present (19)", "all build payload SHA-256 hashes match", "per-artifact quantization manifest", "artifact manifest base model", "artifact manifest source revision", "artifact manifest declares vanilla quantization", "plausible size 30.02 GB in [28, 40]", "required sidecar config.json", "required sidecar tokenizer_config.json", "required sidecar generation_config.json", "required sidecar preprocessor_config.json", "required sidecar video_preprocessor_config.json", "required sidecar chat_template.jinja", "required sidecar tokenizer.json", "required sidecar vocab.json", "required sidecar merges.txt", "required sidecar LICENSE", "chat template byte-identical to source", "generation_config.json semantically intact", "preprocessor_config.json semantically intact", "video_preprocessor_config.json semantically intact", "tokenizer.json byte-identical to source", "vocab.json byte-identical to source", "merges.txt byte-identical to source", "LICENSE byte-identical to source", "tokenizer config preserves chat_template", "tokenizer config preserves eos_token", "tokenizer config preserves pad_token", "tokenizer config preserves additional_special_tokens", "official internal architecture id retained", "MTP layer declaration retained", "vision configuration retained", "image special token id retained", "video special token id retained", "vision-start token id retained", "vision-end token id retained", "text architecture preserves num_hidden_layers", "text architecture preserves hidden_size", "text architecture preserves vocab_size", "text architecture preserves max_position_embeddings", "text architecture preserves full_attention_interval", "text architecture preserves layer_types", "text architecture preserves linear_conv_kernel_dim", "text architecture preserves linear_key_head_dim", "text architecture preserves linear_num_key_heads", "text architecture preserves linear_num_value_heads", "text architecture preserves linear_value_head_dim", "full vision configuration retained", "main safetensors readable", "vision tensors retained (333)", "native MTP tensors retained in oMLX bundle (29)", "all floating tensors finite" ], "failures": [], "warnings": [], "disk_size_bytes": 30024785650 }, "runtime": { "runtime": "oMLX", "model": "/Users/oz/Documents/ChatGPT/qwen3.8-27b/qwen38-27b-quants/quantization-suite/artifacts/MLX-oQ8", "text_tests": [ { "id": "fact", "category": "factual", "passed": true, "failures": [], "text": "The capital of France is Paris.", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 24, "completion_tokens": 7, "total_tokens": 31, "input_tokens": 24, "output_tokens": 7, "prompt_tokens_details": { "cached_tokens": 0 }, "model_load_duration": 3.45, "total_time": 2.79 }, "timings": null, "wall_seconds": 6.269825999857858, "loop_score": 0.0 }, { "id": "math", "category": "mathematical_reasoning", "passed": true, "failures": [], "text": "6 × 7 = **42** blue marbles.", "reasoning_content": "The user asks a simple multiplication problem: 6 rows of 7 blue marbles. I need to show a short calculation.\n\n6 × 7 = 42", "finish_reason": "stop", "usage": { "prompt_tokens": 36, "completion_tokens": 51, "total_tokens": 87, "input_tokens": 36, "output_tokens": 51, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 3.22 }, "timings": null, "wall_seconds": 3.223372000036761, "loop_score": 0.0 }, { "id": "code", "category": "coding", "passed": true, "failures": [], "text": "```python\ndef clamp(x, low, high):\n return max(low, min(x, high))\n```", "reasoning_content": "The user wants a Python function called `clamp(x, low, high)` that clamps a value `x` to be within the inclusive range `[low, high]`. This is a straightforward function:\n- If x < low, return low\n- If x > high, return high\n- Otherwise, return x\n\nI can write this using `max(low, min(x, high))` or with if/elif/else. The one-liner with max/min is clean and Pythonic. Let me provide a single fenced code block.", "finish_reason": "stop", "usage": { "prompt_tokens": 67, "completion_tokens": 138, "total_tokens": 205, "input_tokens": 67, "output_tokens": 138, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 7.92 }, "timings": null, "wall_seconds": 7.928127582883462, "loop_score": 0.0 }, { "id": "multiturn", "category": "multi_turn", "passed": true, "failures": [], "text": "cedar", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 50, "completion_tokens": 2, "total_tokens": 52, "input_tokens": 50, "output_tokens": 2, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 0.76 }, "timings": null, "wall_seconds": 0.7602953750174493, "loop_score": 0.0 }, { "id": "thinking_off", "category": "thinking_disabled", "passed": true, "failures": [], "text": "Yes, 17 is prime.\n\n**Reason:** It has no divisors other than 1 and itself (it is not divisible by 2, 3, or any other prime number less than its square root, $\\sqrt{17} \\approx 4.12$).", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 25, "completion_tokens": 61, "total_tokens": 86, "input_tokens": 25, "output_tokens": 61, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 3.55 }, "timings": null, "wall_seconds": 3.556886708829552, "loop_score": 0.0 }, { "id": "reasoning_low", "category": "reasoning_effort_low", "passed": true, "failures": [], "text": "3/5", "reasoning_content": "Compare 3/5 and 5/9.\n\n3/5 = 27/45\n5/9 = 25/45\n\n27/45 > 25/45, so 3/5 is larger.", "finish_reason": "stop", "usage": { "prompt_tokens": 60, "completion_tokens": 61, "total_tokens": 121, "input_tokens": 60, "output_tokens": 61, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 3.36 }, "timings": null, "wall_seconds": 3.3681672499515116, "loop_score": 0.0 }, { "id": "reasoning_high", "category": "reasoning_effort_high", "passed": true, "failures": [], "text": "The smallest positive integer divisible by 6, 8, and 15 is their least common multiple (LCM).\n\nPrime factorizations:\n- 6 = 2 × 3\n- 8 = 2³\n- 15 = 3 × 5\n\nTake the highest power of each prime:\n- 2³, 3, 5\n\nLCM = 2³ × 3 × 5 = 8 × 3 × 5 = **120**.", "reasoning_content": "We need answer user: Find smallest positive integer divisible by 6,8,15. Explain briefly. Need compute LCM. 6=2*3, 8=2^3, 15=3*5. LCM max powers: 2^3*3*5=120. Need final concise.", "finish_reason": "stop", "usage": { "prompt_tokens": 73, "completion_tokens": 181, "total_tokens": 254, "input_tokens": 73, "output_tokens": 181, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 8.64 }, "timings": null, "wall_seconds": 8.646054749842733, "loop_score": 0.0 }, { "id": "long_form", "category": "long_form", "passed": true, "failures": [], "text": "Checksums serve as the foundational integrity layer for reproducible machine learning by providing cryptographic fingerprints for every artifact in the release pipeline, including raw datasets, preprocessed features, and model weights. By verifying that these hashes match expected values, teams can ensure that the exact data and code versions used during training are identical to those deployed in production, eliminating subtle discrepancies caused by silent data drift or untracked file modifications. This deterministic verification process guarantees that any stakeholder can independently confirm that the software package has not been altered or corrupted since its initial creation.\n\nImmutable revisions extend this integrity to the version control and dependency management layers, ensuring that the specific state of the codebase, library versions, and environment configurations remains fixed and unchangeable over time. Unlike mutable branches that can be updated or rebased, immutable tags or container images lock in the precise combination of source code and dependencies required to reproduce a specific model behavior. This prevents the \"works on my machine\" problem by ensuring that the execution environment is bit-for-bit identical across development, testing, and production stages, thereby isolating the model from external environmental variables.\n\nTogether, these mechanisms create a verifiable audit trail that links every prediction back to its exact origin, which is critical for debugging, compliance, and scientific validity. When a model exhibits unexpected behavior, engineers can trace the issue by comparing the checksums of the deployed artifacts against the original training records, quickly identifying whether the fault lies in the data, the code, or the environment. This transparency transforms machine learning from a black box into a deterministic system where outcomes are fully explainable and traceable to specific, verified inputs.\n\nUltimately, the integration of checksums and immutable revisions reduces the operational risk associated with ML deployments by establishing a standard of trust and consistency. Organizations can confidently scale their models knowing that reproducibility is not merely a theoretical goal but an enforced technical constraint. This reliability accelerates the development lifecycle by allowing teams to iterate rapidly without fear of introducing hidden inconsistencies, while simultaneously satisfying regulatory requirements for data provenance and algorithmic accountability in high-stakes applications.", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 31, "completion_tokens": 416, "total_tokens": 447, "input_tokens": 31, "output_tokens": 416, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 28.17 }, "timings": null, "wall_seconds": 28.175676584010944, "loop_score": 0.0 }, { "id": "non_english", "category": "non_english", "passed": true, "failures": [], "text": "太陽光が大気中の分子にぶつかると、波長の短い青い光が他の色よりも強く散乱されます。そのため、私たちの目には空全体が青く見えます。", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 28, "completion_tokens": 40, "total_tokens": 68, "input_tokens": 28, "output_tokens": 40, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 3.73 }, "timings": null, "wall_seconds": 3.735479042166844, "loop_score": 0.0 }, { "id": "json", "category": "structured_json", "passed": true, "failures": [], "text": "{\n \"name\": \"sample\",\n \"primes\": [2, 3, 5],\n \"active\": true\n}", "reasoning_content": null, "finish_reason": "stop", "usage": { "prompt_tokens": 41, "completion_tokens": 32, "total_tokens": 73, "input_tokens": 41, "output_tokens": 32, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 2.14 }, "timings": null, "wall_seconds": 2.147881958168, "loop_score": 0.0 } ], "tool_tests": [ { "id": "one_tool", "passed": true, "failures": [], "calls": [ { "id": "call_05171591", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null }, { "id": "parallel_tools", "passed": true, "failures": [], "calls": [ { "id": "call_a7a2ed35", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } }, { "id": "call_624052ee", "name": "weather_at", "arguments": { "location": "Austin", "units": "fahrenheit" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null }, { "id": "sequential_dependency", "passed": true, "failures": [], "calls": [ { "id": "call_c4475109", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": "" }, { "id": "nested_json", "passed": true, "failures": [], "calls": [ { "id": "call_5c5dad46", "name": "add_event", "arguments": { "title": "Release review", "attendees": [ "ana@example.test", "li@example.test" ], "metadata": { "release": { "model": "Qwen3.8-27B", "slot": 21 } } } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null }, { "id": "invalid_recovery", "passed": true, "failures": [], "calls": [ { "id": "call_1ce66614", "name": "weather_at", "arguments": { "location": "Paris", "units": "celsius" } } ], "raw": "", "finish_reason": "tool_calls", "dependent_raw": null } ], "vision_tests": [ { "id": "shapes_colors", "passed": true, "missing_patterns": [], "text": "From left to right, the three large shapes are:\n\n1. **Red square** \n2. **Blue circle** \n3. **Green triangle**\n\nThese match the labels beneath each shape in the image: “RED”, “BLUE”, and “GREEN” respectively.", "finish_reason": "stop" }, { "id": "printed_text", "passed": true, "missing_patterns": [], "text": "VISION CHECK 27B", "finish_reason": "stop" }, { "id": "chart", "passed": true, "missing_patterns": [], "text": "Based on the provided image:\n\n- The bar chart has three bars labeled **A**, **B**, and **C**.\n- The numbers printed above each bar are:\n - **A**: 60\n - **B**: 105\n - **C**: 135\n\nThe tallest bar is **C**, as it reaches the highest point on the y-axis, and the number printed above it is **135**.\n\n✅ **Answer: Bar C is the tallest, and the number printed above it is 135.**", "finish_reason": "stop" } ], "mtp": { "passed": true, "backend": "oMLX Lightning MTP (integrated source MTP head)", "output_equivalent_temperature_zero": true, "baseline": { "text": "Here are the first ten square numbers, listed with commas:\n\n1. 1\n2. 4\n3. 9\n4. 16\n5. 25\n6. 36\n7. 49\n8. 64\n9. 81\n10. 100", "finish_reason": "stop", "usage": { "prompt_tokens": 21, "completion_tokens": 71, "total_tokens": 92, "input_tokens": 21, "output_tokens": 71, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 8.91 }, "wall_seconds": 8.91301712510176, "generation_tps": 7.968574635241302 }, "mtp_measurement": { "text": "Here are the first ten square numbers, listed with commas:\n\n1. 1\n2. 4\n3. 9\n4. 16\n5. 25\n6. 36\n7. 49\n8. 64\n9. 81\n10. 100", "finish_reason": "stop", "usage": { "prompt_tokens": 21, "completion_tokens": 71, "total_tokens": 92, "input_tokens": 21, "output_tokens": 71, "prompt_tokens_details": { "cached_tokens": 0 }, "total_time": 3.68 }, "wall_seconds": 3.687670208979398, "generation_tps": 19.293478260869563 }, "baseline_tps": 7.968574635241302, "mtp_tps": 19.293478260869563, "speedup": 2.4211956521739126, "measured_improvement": true, "advertise_acceleration": true, "native_stats": { "finish_reason": "stop", "tokens": 72, "cycles": 21, "tokens_per_cycle": 3.43, "accepted_drafts": 51, "drafted_tokens": 51, "acceptance_rate": 1.0 }, "failure": null }, "peak_process_rss_gb": 27.20964608, "validation_inputs": { "prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528", "tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122", "image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295" }, "warnings": [ "Peak process RSS is a unified-memory process measurement, not a separate VRAM measurement." ] }, "warnings": [], "overall_passed": true, "runtime_failures": [], "quality": { "schema_version": 1, "comparison_type": "cross-runtime output agreement against pinned BF16 source", "passed": true, "source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "embedding_model": { "repo_id": "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2", "revision": "e8f8c211226b894fcb81acc59f3b34ba3efd5f42", "pooling": "attention-mask mean pooling followed by L2 normalization", "maximum_tokens": 256 }, "thresholds": { "mean_semantic_similarity": 0.55, "per_case_severe_regression": 0.25 }, "baseline_valid": true, "candidate_functional": true, "semantic_gate_passed": true, "validation_inputs_match": true, "validation_inputs": { "prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528", "tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122", "image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295" }, "mean_semantic_similarity": 0.8972687065601349, "exact_matches": 4, "comparisons": [ { "id": "fact", "reference_passed": true, "candidate_passed": true, "exact_match": true, "sequence_agreement": 1.0, "semantic_similarity": 1.0, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "math", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.2994652406417112, "semantic_similarity": 0.8161226511001587, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "code", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.25622775800711745, "semantic_similarity": 0.867202639579773, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "multiturn", "reference_passed": true, "candidate_passed": true, "exact_match": true, "sequence_agreement": 1.0, "semantic_similarity": 1.0, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "thinking_off", "reference_passed": true, "candidate_passed": true, "exact_match": true, "sequence_agreement": 1.0, "semantic_similarity": 0.9999998807907104, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "reasoning_low", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.0625, "semantic_similarity": 0.6260854005813599, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "reasoning_high", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.7183308494783904, "semantic_similarity": 0.8893308639526367, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "long_form", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.20647693817468105, "semantic_similarity": 0.8738414645195007, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "non_english", "reference_passed": true, "candidate_passed": true, "exact_match": true, "sequence_agreement": 1.0, "semantic_similarity": 1.0000001192092896, "severe_regression": false, "candidate_loop_score": 0.0 }, { "id": "json", "reference_passed": true, "candidate_passed": true, "exact_match": false, "sequence_agreement": 0.9130434782608695, "semantic_similarity": 0.9001040458679199, "severe_regression": false, "candidate_loop_score": 0.0 } ], "functional_results": { "reference_text": { "passed": 10, "total": 10 }, "candidate_text": { "passed": 10, "total": 10 }, "reference_tools": { "passed": 5, "total": 5 }, "candidate_tools": { "passed": 5, "total": 5 }, "reference_vision": { "passed": 3, "total": 3 }, "candidate_vision": { "passed": 3, "total": 3 } }, "measurements": { "average_generation_tps": 13.513484058818289, "peak_memory_gb": 27.20964608, "artifact_bytes": 30024788604, "maximum_prompt_tokens_tested": 73, "loop_rate": 0.0 }, "warnings": [ "Semantic similarity is a measured embedding-model proxy, not ground-truth accuracy.", "Raw-logit equality is unavailable across all target runtimes; exact functional gates and output agreement are used for portable release validation.", "Sequence agreement is lexical and is reported diagnostically, not used as semantic accuracy." ] }, "quality_failures": [] }