{ "aggregate_validation_sha256": "e81ab89c81cbe6b3036e4eb83aff713ee6bf214c41a9efa0c31d044bb1396223", "byte_identical_to_validated_aggregate": true, "mtp": { "acceleration_claimed": false, "advertise_acceleration": false, "passed": true, "retention_gate": "GGUF tensor and nextn metadata inspection" }, "overall_passed": true, "per_file_runtime_smoke": { "file": "Qwen3.8-27B-Q3_K_S.gguf", "passed": true, "text": "Paris", "timings": { "cache_n": 0, "predicted_ms": 57.64, "predicted_n": 2, "predicted_per_second": 17.349063150589867, "predicted_per_token_ms": 57.64, "prompt_ms": 379.479, "prompt_n": 25, "prompt_per_second": 65.8797983551132, "prompt_per_token_ms": 15.17916 } }, "quality_passed": true, "schema_version": 1, "source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "target": "GGUF-Q3_K_S", "text_tests": [ { "category": "factual", "failures": [], "finish_reason": "stop", "id": "fact", "loop_score": 0.0, "passed": true, "reasoning_content": null, "text": "The capital of France is Paris.", "timings": { "cache_n": 0, "predicted_ms": 828.35, "predicted_n": 8, "predicted_per_second": 8.450534194483009, "predicted_per_token_ms": 118.33571428571429, "prompt_ms": 407.795, "prompt_n": 24, "prompt_per_second": 58.85310020966416, "prompt_per_token_ms": 16.991458333333334 }, "usage": { "completion_tokens": 8, "prompt_tokens": 24, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 32 }, "wall_seconds": 1.2432157919974998 }, { "category": "mathematical_reasoning", "failures": [], "finish_reason": "stop", "id": "math", "loop_score": 0.0, "passed": true, "reasoning_content": "The user asks a simple multiplication problem: 6 rows of 7 blue marbles. I need to show a short calculation.\n\n6 \u00d7 7 = 42\n", "text": "6 \u00d7 7 = **42** blue marbles.", "timings": { "cache_n": 0, "predicted_ms": 6022.075, "predicted_n": 52, "predicted_per_second": 8.468841719839093, "predicted_per_token_ms": 118.07990196078431, "prompt_ms": 401.913, "prompt_n": 36, "prompt_per_second": 89.5716237096088, "prompt_per_token_ms": 11.164250000000001 }, "usage": { "completion_tokens": 52, "prompt_tokens": 36, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 88 }, "wall_seconds": 6.449553958955221 }, { "category": "coding", "failures": [], "finish_reason": "stop", "id": "code", "loop_score": 0.0, "passed": true, "reasoning_content": "The user wants a Python function called `clamp(x, low, high)` that clamps a value `x` to be within the inclusive range `[low, high]`. This is a straightforward function:\n- If x < low, return low\n- If x > high, return high\n- Otherwise, return x\n\nI can write this using `max(low, min(x, high))` or with if/elif/else. The one-liner with max/min is clean and Pythonic. Let me provide a single fenced code block.\n", "text": "```python\ndef clamp(x, low, high):\n return max(low, min(x, high))\n```", "timings": { "cache_n": 0, "predicted_ms": 16177.57, "predicted_n": 139, "predicted_per_second": 8.530329338707853, "predicted_per_token_ms": 117.22876811594203, "prompt_ms": 681.445, "prompt_n": 67, "prompt_per_second": 98.32048074312674, "prompt_per_token_ms": 10.17082089552239 }, "usage": { "completion_tokens": 139, "prompt_tokens": 67, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 206 }, "wall_seconds": 16.879085917025805 }, { "category": "multi_turn", "failures": [], "finish_reason": "stop", "id": "multiturn", "loop_score": 0.0, "passed": true, "reasoning_content": null, "text": "cedar", "timings": { "cache_n": 0, "predicted_ms": 242.343, "predicted_n": 3, "predicted_per_second": 8.25276570810793, "predicted_per_token_ms": 121.1715, "prompt_ms": 662.359, "prompt_n": 50, "prompt_per_second": 75.4877641883027, "prompt_per_token_ms": 13.24718 }, "usage": { "completion_tokens": 3, "prompt_tokens": 50, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 53 }, "wall_seconds": 0.9361831250134856 }, { "category": "thinking_disabled", "failures": [], "finish_reason": "stop", "id": "thinking_off", "loop_score": 0.0, "passed": true, "reasoning_content": null, "text": "Yes, 17 is prime.\n\n**Reason:** It has no divisors other than 1 and itself (it is not divisible by 2, 3, or any other prime number less than its square root, $\\sqrt{17} \\approx 4.12$).", "timings": { "cache_n": 0, "predicted_ms": 7006.669, "predicted_n": 62, "predicted_per_second": 8.705991391915331, "predicted_per_token_ms": 114.86342622950819, "prompt_ms": 394.813, "prompt_n": 25, "prompt_per_second": 63.32111657924131, "prompt_per_token_ms": 15.79252 }, "usage": { "completion_tokens": 62, "prompt_tokens": 25, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 87 }, "wall_seconds": 7.437449707998894 }, { "category": "reasoning_effort_low", "failures": [], "finish_reason": "stop", "id": "reasoning_low", "loop_score": 0.0, "passed": true, "reasoning_content": "Compare 3/5 and 5/9.\n\n3/5 = 27/45\n5/9 = 25/45\n\n27/45 > 25/45, so 3/5 is larger.\n", "text": "3/5", "timings": { "cache_n": 0, "predicted_ms": 7045.161, "predicted_n": 62, "predicted_per_second": 8.658425265228148, "predicted_per_token_ms": 115.49444262295081, "prompt_ms": 670.127, "prompt_n": 60, "prompt_per_second": 89.53526719562115, "prompt_per_token_ms": 11.168783333333332 }, "usage": { "completion_tokens": 62, "prompt_tokens": 60, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 122 }, "wall_seconds": 7.7370391250588 }, { "category": "reasoning_effort_high", "failures": [], "finish_reason": "stop", "id": "reasoning_high", "loop_score": 0.0, "passed": true, "reasoning_content": "We need answer user: Find smallest positive integer divisible by 6,8,15. Explain briefly. Need compute LCM. 6=2*3, 8=2^3, 15=3*5. LCM max powers: 2^3*3*5=120. Need final concise.\n", "text": "The smallest positive integer divisible by 6, 8, and 15 is their least common multiple (LCM).\n\nPrime factorizations:\n- 6 = 2 \u00d7 3\n- 8 = 2\u00b3\n- 15 = 3 \u00d7 5\n\nTake the highest power of each prime:\n- 2\u00b3, 3, 5\n\nLCM = 2\u00b3 \u00d7 3 \u00d7 5 = 8 \u00d7 3 \u00d7 5 = **120**.", "timings": { "cache_n": 0, "predicted_ms": 21175.912, "predicted_n": 182, "predicted_per_second": 8.54744768489782, "predicted_per_token_ms": 116.99398895027625, "prompt_ms": 674.667, "prompt_n": 73, "prompt_per_second": 108.20152756841523, "prompt_per_token_ms": 9.242013698630137 }, "usage": { "completion_tokens": 182, "prompt_tokens": 73, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 255 }, "wall_seconds": 21.88782295899 }, { "category": "long_form", "failures": [], "finish_reason": "stop", "id": "long_form", "loop_score": 0.0, "passed": true, "reasoning_content": null, "text": "Checksums serve as the foundational integrity layer for reproducible machine learning by providing cryptographic fingerprints for every artifact in the release pipeline, including raw datasets, preprocessed features, and model weights. By verifying that these hashes match expected values, engineers can guarantee that the exact same input data and code versions are used during training and inference, eliminating subtle discrepancies caused by silent data drift or untracked file modifications. This deterministic verification ensures that any team member or automated system can reconstruct the identical environment, thereby establishing a trusted baseline for all subsequent experiments and deployments.\n\nImmutable revisions extend this reliability by locking the entire project state\u2014including source code, dependency versions, and configuration files\u2014into a permanent, unchangeable snapshot. Unlike mutable branches that can be altered after the fact, an immutable revision acts as a historical record that cannot be tampered with, ensuring that the specific combination of libraries and hyperparameters used to generate a model is preserved exactly as it was at the time of creation. This prevents the \"works on my machine\" problem by decoupling the release from the evolving state of the development environment, allowing for precise auditing and debugging of past performance metrics.\n\nTogether, these mechanisms create a closed loop of trust where the checksum validates the content of the immutable revision, ensuring that the locked state contains the exact binary and data assets required for reproduction. When a model is deployed, the system can automatically verify that the runtime environment matches the checksums defined in the immutable release manifest, instantly flagging any deviations that might lead to non-deterministic behavior or security vulnerabilities. This synergy transforms ML releases from opaque black boxes into transparent, verifiable packages where every component\u2019s origin and integrity are mathematically provable.\n\nUltimately, the integration of checksums and immutable revisions significantly reduces the operational risk and maintenance overhead associated with scaling machine learning systems. By enabling rapid rollback to known-good states and facilitating seamless handoffs between data scientists and MLOps engineers, these practices foster a culture of accountability and precision. As a result, organizations can confidently iterate on models while maintaining strict compliance and reproducibility, ensuring that every deployed model behaves consistently across different environments and over time.", "timings": { "cache_n": 0, "predicted_ms": 49899.86, "predicted_n": 431, "predicted_per_second": 8.617258645615438, "predicted_per_token_ms": 116.04618604651164, "prompt_ms": 401.41, "prompt_n": 31, "prompt_per_second": 77.22777210333572, "prompt_per_token_ms": 12.948709677419355 }, "usage": { "completion_tokens": 431, "prompt_tokens": 31, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 462 }, "wall_seconds": 50.34475950000342 }, { "category": "non_english", "failures": [], "finish_reason": "stop", "id": "non_english", "loop_score": 0.0, "passed": true, "reasoning_content": null, "text": "\u592a\u967d\u5149\u304c\u5927\u6c17\u4e2d\u306e\u5206\u5b50\u306b\u3076\u3064\u304b\u308b\u3068\u3001\u6ce2\u9577\u306e\u77ed\u3044\u9752\u3044\u5149\u304c\u4ed6\u306e\u8272\u3088\u308a\u3082\u5f37\u304f\u6563\u4e71\u3055\u308c\u307e\u3059\u3002\u305d\u306e\u305f\u3081\u3001\u79c1\u305f\u3061\u306e\u76ee\u306b\u306f\u7a7a\u5168\u4f53\u304c\u9752\u304f\u898b\u3048\u307e\u3059\u3002", "timings": { "cache_n": 0, "predicted_ms": 4555.149, "predicted_n": 41, "predicted_per_second": 8.781271479813283, "predicted_per_token_ms": 113.878725, "prompt_ms": 399.997, "prompt_n": 28, "prompt_per_second": 70.00052500393753, "prompt_per_token_ms": 14.285607142857144 }, "usage": { "completion_tokens": 41, "prompt_tokens": 28, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 69 }, "wall_seconds": 4.98401399992872 }, { "category": "structured_json", "failures": [], "finish_reason": "stop", "id": "json", "loop_score": 0.0, "passed": true, "reasoning_content": null, "text": "```json\n{\n \"name\": \"sample\",\n \"primes\": [2, 3, 5],\n \"active\": true\n}\n```", "timings": { "cache_n": 0, "predicted_ms": 4116.45, "predicted_n": 38, "predicted_per_second": 8.98832732087114, "predicted_per_token_ms": 111.2554054054054, "prompt_ms": 409.629, "prompt_n": 41, "prompt_per_second": 100.09056975946527, "prompt_per_token_ms": 9.990951219512196 }, "usage": { "completion_tokens": 38, "prompt_tokens": 41, "prompt_tokens_details": { "cached_tokens": 0 }, "total_tokens": 79 }, "wall_seconds": 4.555776666966267 } ], "tool_tests": [ { "calls": [ { "arguments": { "location": "Paris", "units": "celsius" }, "id": "0ZODzhzvv18YFnouB8e35NCBQwdKSddU", "name": "weather_at" } ], "dependent_raw": null, "failures": [], "finish_reason": "tool_calls", "id": "one_tool", "passed": true, "raw": "" }, { "calls": [ { "arguments": { "location": "Paris", "units": "celsius" }, "id": "0UKRAoNXPGJluje2sWd7Yy7DlTQj9WRa", "name": "weather_at" }, { "arguments": { "location": "Austin", "units": "fahrenheit" }, "id": "gC0y9t7Cq7zYea30NsoeaQOA5F1kK5xs", "name": "weather_at" } ], "dependent_raw": null, "failures": [], "finish_reason": "tool_calls", "id": "parallel_tools", "passed": true, "raw": "" }, { "calls": [ { "arguments": { "location": "Paris", "units": "celsius" }, "id": "sHYKcUM2PmfJbC6Xdy9UjD66y0E6Krlm", "name": "weather_at" } ], "dependent_raw": "", "failures": [], "finish_reason": "tool_calls", "id": "sequential_dependency", "passed": true, "raw": "" }, { "calls": [ { "arguments": { "attendees": [ "ana@example.test", "li@example.test" ], "metadata": { "release": { "model": "Qwen3.8-27B", "slot": 21 } }, "title": "Release review" }, "id": "8UCZa3yioDsCyTo0SkkRqSMonT6UuHOZ", "name": "add_event" } ], "dependent_raw": null, "failures": [], "finish_reason": "tool_calls", "id": "nested_json", "passed": true, "raw": "" }, { "calls": [ { "arguments": { "location": "Paris", "units": "celsius" }, "id": "TmbG56wrC2DvxJNr6gF15IlcZqFjru9H", "name": "weather_at" } ], "dependent_raw": null, "failures": [], "finish_reason": "tool_calls", "id": "invalid_recovery", "passed": true, "raw": "" } ], "validation_inputs": { "image_sha256": "0b1ae6badbe19a6049305c36d165a34cdf36df00033842e265339b2b7f057295", "prompts_sha256": "136a918e5fee962f2b52f8e520a0275fd5ec5569181e0f5fdf4910ee3c34d528", "tools_sha256": "86ae46ebdbb0324c9672eec87e6b7f7683b0eb9c8c7169dcabdf646cf9bab122" }, "vision_tests": [ { "finish_reason": "stop", "id": "shapes_colors", "missing_patterns": [], "passed": true, "text": "From left to right, the three large shapes are:\n\n1. **Red square** \n2. **Blue circle** \n3. **Green triangle**\n\nThese match the labels directly beneath each shape in the image: \u201cRED\u201d, \u201cBLUE\u201d, and \u201cGREEN\u201d respectively." }, { "finish_reason": "stop", "id": "printed_text", "missing_patterns": [], "passed": true, "text": "VISION CHECK 27B" }, { "finish_reason": "stop", "id": "chart", "missing_patterns": [], "passed": true, "text": "Based on the provided image:\n\n- The bar chart has three bars labeled **A**, **B**, and **C**.\n- The numbers printed above each bar are:\n - **A**: 60\n - **B**: 105\n - **C**: 135\n\nThe tallest bar is **C**, as it reaches the highest point on the y-axis, and the number printed above it is **135**.\n\n\u2705 **Answer: Bar C is the tallest, and the number printed above it is 135.**" } ] }