{ "attention_projections_full_precision": true, "calibration": { "calls": 3934, "conversations": 3, "effective_rows_lower_bound": 3934, "input": "captured exact combined_embeds; no tokenization", "manifest_sha256": "8522310fb27db4ebe075628b720058693377a8cc902e1155187f38e2bd02efb8", "selection_manifest_sha256": "4d38a0f3384a2f0695a60f42ad22c5945be6c0b094d5628779ca1776f4ad3402" }, "embeddings_full_precision": true, "evaluation": { "calls": 1172, "conversations": 1, "disjoint_from_calibration": true, "manifest_sha256": "2080deb829fcb9cd6fc524edc0dca86cb591af1d9d8abfd10dbc9632e15b1a6b", "selection_manifest_sha256": "4d38a0f3384a2f0695a60f42ad22c5945be6c0b094d5628779ca1776f4ad3402" }, "function_head_full_precision": true, "kind": "portable_nano_conversion_report", "layer_metrics": [], "method": "calibrated symmetric GPTQ MLP+Mamba+text-head W8A16 group-128; K=15680 MLP-down padded to K=15744; function head and embeddings full precision", "policy": "w8-all-g128", "quantization_config": { "bits": 8, "checkpoint_format": "gptq", "desc_act": false, "dynamic": { "-:.*(?:embed_tokens|embed_asr_tokens)$": {}, "-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}, "-:.*function_head$": {} }, "group_size": 128, "lm_head": true, "modules_in_block_to_quantize": [ "mixer.up_proj", "mixer.down_proj", "mixer.in_proj", "mixer.out_proj", "lm_head" ], "quant_method": "gptq", "sym": true }, "quantized_output_heads": [ "lm_head" ], "reproducibility": { "byte_identical": true, "runs": 2, "tree_sha256": "0d7c0d8e8503ece71b2b606c86dad371c3d7fc5b58d892945068b14331029bb3" }, "runtime": { "driver_version": "580.142", "kernel_version": "6.17.0-1014-nvidia", "packages": { "compressed-tensors": "0.13.0", "nvidia-modelopt": "0.37.0", "safetensors": "0.8.0", "torch": "2.10.0a0+b4e4ee81d3.nv25.12", "transformers": "4.56.0", "vllm": "0.17.1.dev0+gb31e9326a.fi065" } }, "schema": 1, "selected_source_bytes": 32314490880, "selected_tensor_count": 105, "shape_exception": { "effective_group_size": 128, "other_group_size_fallbacks_allowed": false, "padded_shape": [ 4480, 15744 ], "reason": "zero-pad K by 64 so all 105 selected tensors remain group-128", "requested_group_size": 128, "runtime_padding": true, "selector": "stt_model.llm.layers.*.mixer.down_proj.weight", "shape": [ 4480, 15680 ] }, "source_provenance": { "checkpoint_sha256": "d553750c29434a6bb524377e17634c6cafdbf621892e643a77f406e51570354b", "derived_from_ea": false, "nano_sha256": "1b473452103362817e9408cf864aa8ef8ca92a48ef8fd94cb3d40fa7cf14e44b", "repository": "nvidia/NVIDIA-NemotronLabs-VoiceChat-11B", "revision": "fb0f94eaf4d03ddc430f39565229393fa1b50c26" }, "status": "complete" }