{ "schema_version": "0.1.0", "release": { "id": "Qwen3.8-27B-MixQ4A6E8-v0-rc1", "display_name": "Qwen3.8-27B MixQ4A6E8-v0 experimental candidate rc1", "status": "candidate", "candidate_class": "experimental", "created_utc": "2026-08-19T13:20:00Z", "maintainer": null, "immutable": true, "supersedes": null, "artifact_status": "built", "experimental_validation_status": "complete_for_declared_v0_scope" }, "source": { "repo_id": "Qwen/Qwen3.8-27B", "revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "architecture": "qwen35", "parameter_count": 27300000000, "license": { "spdx": "Apache-2.0", "source_url": "https://huggingface.co/Qwen/Qwen3.8-27B/blob/1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0/LICENSE", "verified": true }, "files": [ { "role": "quantization_input", "repo_id": "ggml-org/Qwen3.8-27B-GGUF", "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf", "filename": "Qwen3.8-27B-BF16.gguf", "size_bytes": 53808281952, "sha256": "5a3eedc837bcbd1365cdbf5b71e698df3122e76586ba872f07ce3ed4a9bfa97e" } ] }, "conversion": { "kind": "gguf_requantization", "repository": "https://github.com/ggml-org/llama.cpp", "release_tag": "b10442", "revision": "9b0a2ce859d3884252705e9ac7c93c98616bb238", "binary_asset": "llama-b10442-bin-ubuntu-vulkan-x64.tar.gz", "binary_asset_size_bytes": 33235772, "binary_asset_sha256": "cb3e21b9659c8772cfd1ba5c1fdc76f98a60ee4864d5b314844cdf0b16f91796", "commands_manifest": "manifests/candidate-MixQ4A6E8-v0.json", "host": "dual-rtx3090-validation-host" }, "quantization": { "profile_id": "MixQ4A6E8-v0", "status": "experimental_built", "objective": "Evaluate a Q4_K_M fallback with Q6_K attention/SSM matches and Q8_0 embedding/output tensors for single-24GB-GPU use; no optimization outcome is asserted.", "fallback_type": "Q4_K_M", "tensor_rules": [ { "pattern": ".*attn.*", "type": "Q6_K" }, { "pattern": ".*ssm.*", "type": "Q6_K" } ], "token_embedding_type": "Q8_0", "output_tensor_type": "Q8_0", "tensor_rules_path": "config/MixQ4A6E8-v0.tensor-rules.json", "tensor_rules_sha256": "4000c7ce09babd418f533a97800f3052a970a547e49d51eb92a349b5450c40b7", "calibration": { "provenance": "external_bartowski1182", "repo_id": "bartowski/Qwen3.8-27B-GGUF", "revision": "f0eec4a4bb4975114a030d048952d83c0a53c034", "corpus_filename": "Qwen3.8-27B-calibration-v6.txt", "corpus_sha256": "b3e9d2371c8931cafa5b6184adcee5cb59adb7f4103e28d0f99eddac1cced4e6", "imatrix_filename": "Qwen3.8-27B-imatrix.gguf", "imatrix_size_bytes": 13642688, "imatrix_sha256": "aaa933d4b9ce23e1f65c548ad34f16956d8af44a51b5c15bf4f393ba59508cd8" } }, "artifacts": [ { "id": "main-model", "role": "main_model", "filename": "Qwen3.8-27B-MixQ4A6E8-v0.gguf", "size_bytes": 19021793888, "sha256": "5cd87bedac2cec27d2b16203ef35b4fa861cf3a6e409a28a7133b37237e49509", "bundled_in_git": false, "upload": true } ], "companion_artifacts": [ { "role": "multimodal_projector", "quantization": "Q8_0", "filename": "mmproj-Qwen3.8-27B-Q8_0.gguf", "size_bytes": 629247008, "sha256": "2e968a6af97ce35d8971890b257b9b7edabf20ad91450501fa53162a19ee33eb", "source_manifest": "manifests/companions-ggml-org-0669b986.json", "bundled": false, "upload": false, "status": "separate_unbundled_experimental_failed_runtime_gate" }, { "role": "mtp_draft_model", "quantization": "Q4_0", "filename": "mtp-Qwen3.8-27B-Q4_0.gguf", "size_bytes": 1680271648, "sha256": "051a1764cff8c4f3ee6ae8b00593a0364c7539c67fa50ffc58f3f96509fca38e", "source_manifest": "manifests/companions-ggml-org-0669b986.json", "bundled": false, "upload": false, "status": "separate_unbundled_experimental" }, { "role": "mtp_draft_model", "quantization": "Q8_0", "filename": "mtp-Qwen3.8-27B-Q8_0.gguf", "size_bytes": 3164006688, "sha256": "cbf60a0c48b431bb61f1d49b8948dc88ac29c398d6dbdbbb2e6e89ef77eacc9a", "source_manifest": "manifests/companions-ggml-org-0669b986.json", "bundled": false, "upload": false, "status": "separate_unbundled_experimental" } ], "validation": { "protocol_version": "0.2.1", "required_hardware": "1-2 x NVIDIA RTX 3090 24 GiB (evidence covers both single-GPU and dual-GPU lanes)", "benchmark_run_ids": [ "results/build-mixq4a6e8-v0-20260818T231742Z", "results/ppl-full-bf16ngl48-mixq4a6e8-v0-vulkan-b10442-20260819T063633Z", "results/llama-bench-mixq4a6e8-v0-single-gpu-cuda-b10442-20260818T233255Z", "results/llama-bench-frontier-q4toq8-cuda-b10442-20260819T094256Z", "results/text-validation-mixq4a6e8-v0-vulkan-b10442-20260819T131943Z" ], "required_checks_passed": true, "experimental_checks_recorded": true, "text": "custom_smoke_passed_not_quality_validation", "mtp": "not_run_out_of_scope_for_v0", "multimodal": "not_run_out_of_scope_for_v0", "claims": [ { "kind": "measured_pareto", "statement": "Within the sealed six-model Q4_K_M/MixQ4A6E8-v0/Q5_K_M/MixQ5A6E8-v0/Q6_K/Q8_0 CUDA comparison only, MixQ4A6E8-v0 is a measured intermediate, nondominated point between the Q4_K_M and Q5_K_M tiers across artifact size, BF16-relative KL, and llama-bench pp512, pp2048, and tg128; this narrow result is not a broad optimization or recommendation claim.", "evidence_run_ids": [ "results/llama-bench-frontier-q4toq8-cuda-b10442-20260819T094256Z", "results/ppl-full-bf16ngl48-mixq4a6e8-v0-vulkan-b10442-20260819T063633Z" ] }, { "kind": "measured", "statement": "MixQ4A6E8-v0 shows 18.2492% lower mean BF16-relative KL divergence than the Q4_K_M reference (0.007835 +/- 0.00005 vs 0.009584 +/- 0.00008) and slightly lower full-corpus WikiText-2 perplexity (6.6830 vs 6.6921) on the sealed full lane; the perplexity difference is small and no broad superiority claim is made.", "evidence_run_ids": [ "results/ppl-full-bf16ngl48-mixq4a6e8-v0-vulkan-b10442-20260819T063633Z" ] }, { "kind": "measured", "statement": "MixQ4A6E8-v0 is slower than Q4_K_M: -2.74% pp512, -4.41% pp2048, -4.66% tg128 (median, dual-RTX-3090 CUDA) and -2.37% pp512, -2.32% pp2048, -4.89% tg128 (median, single-RTX-3090 CUDA). It is faster than Q5_K_M on all three dual-GPU workloads (+1.07% pp512, +1.54% pp2048, +2.52% tg128).", "evidence_run_ids": [ "results/llama-bench-frontier-q4toq8-cuda-b10442-20260819T094256Z", "results/llama-bench-mixq4a6e8-v0-single-gpu-cuda-b10442-20260818T233255Z" ] } ], "known_limitations": [ "This is an experimental candidate, not an optimized, recommended, or generally validated release.", "The measured intermediate/nondominated classification is limited to the sealed six-model Q4-to-Q8 CUDA comparison and its declared dimensions; no broad speed, memory/VRAM, PPL-superiority, Q8/BF16-equivalence, or maximum-context claim is made.", "MTP speculative decoding and multimodal projector were not validated for this artifact (out of scope for v0); they are unvalidated here regardless of their status on the sibling MixQ5A6E8-v0 candidate.", "The fixed structural and text-integration smoke suite passed for the candidate (exact tokenizer/architecture metadata parity plus five deterministic integration tests); it is a reproducibility smoke, not an industry quality benchmark.", "All measurements come from one host (beta1, 1-2 x RTX 3090); no other hardware or runtime is covered.", "Companion MTP and projector files are separate and unbundled." ], "measured_q4_q8_frontier": { "scope": "sealed_q4_k_m_to_q8_0_cuda_comparison_only", "classification": "measured_intermediate_nondominated", "dimensions": { "size_bytes": "lower_is_better", "bf16_relative_kl": "lower_is_better", "llama_bench_pp512_median_tps": "higher_is_better", "llama_bench_pp2048_median_tps": "higher_is_better", "llama_bench_tg128_median_tps": "higher_is_better" }, "points": { "Q4_K_M": { "size_bytes": 17772537440, "bf16_relative_kl": 0.009584, "llama_bench_pp512_median_tps": 1406.9, "llama_bench_pp2048_median_tps": 2103.7, "llama_bench_tg128_median_tps": 42.47 }, "MixQ4A6E8-v0": { "size_bytes": 19021793888, "bf16_relative_kl": 0.007835, "llama_bench_pp512_median_tps": 1368.37, "llama_bench_pp2048_median_tps": 2011.01, "llama_bench_tg128_median_tps": 40.49 }, "Q5_K_M": { "size_bytes": 19231099520, "bf16_relative_kl": 0.003949, "llama_bench_pp512_median_tps": 1353.92, "llama_bench_pp2048_median_tps": 1980.47, "llama_bench_tg128_median_tps": 39.49 }, "MixQ5A6E8-v0": { "size_bytes": 20804373120, "bf16_relative_kl": 0.003083, "llama_bench_pp512_median_tps": 1336.72, "llama_bench_pp2048_median_tps": 1963.61, "llama_bench_tg128_median_tps": 37.43 }, "Q6_K": { "size_bytes": 22082529920, "bf16_relative_kl": 0.002099, "llama_bench_pp512_median_tps": 1251.84, "llama_bench_pp2048_median_tps": 1816.35, "llama_bench_tg128_median_tps": 34.56 }, "Q8_0": { "size_bytes": 29116388960, "bf16_relative_kl": null, "llama_bench_pp512_median_tps": 1515.64, "llama_bench_pp2048_median_tps": 2186.57, "llama_bench_tg128_median_tps": 28.58 } }, "candidate_vs_q4_k_m": { "size_percent_larger": 7.0291, "bf16_relative_kl_percent_lower": 18.2492, "llama_bench_percent_change": { "pp512": -2.7385, "pp2048": -4.4062, "tg128": -4.6621 }, "full_ppl": { "candidate": 6.683, "reference": 6.6921, "interpretation": "candidate_lower_difference_small_no_broad_superiority_claim" } }, "statement": "Within the sealed six-model Q4-to-Q8 CUDA comparison only, MixQ4A6E8-v0 is a measured intermediate, nondominated point between the Q4_K_M and Q5_K_M tiers across artifact size, BF16-relative KL, and llama-bench pp512, pp2048, and tg128; this narrow result is not a broad optimization or recommendation claim.", "evidence_run_ids": [ "results/llama-bench-frontier-q4toq8-cuda-b10442-20260819T094256Z", "results/ppl-full-bf16ngl48-mixq4a6e8-v0-vulkan-b10442-20260819T063633Z" ] } }, "publication": { "target_repo": "Alogotron/Qwen3.8-27B-MixQ4A6E8-GGUF", "visibility": "draft", "model_card": "README.md", "files": [ "Qwen3.8-27B-MixQ4A6E8-v0.gguf", "README.md", "MANIFEST.json", "SHA256SUMS", "LICENSE", "NOTICE", "ATTRIBUTION.md", "UPLOAD.json" ], "upload_manifest": "UPLOAD.json", "companion_policy": "not_included_upload_separately_from_pinned_upstream_if_needed", "published_revision": null, "published_utc": null } }