fhw8: quantize function_head to GPTQ-W8 (production-candidate-3-fhw8)
Browse filesDeterministic RTN requantization of the F32 function_head onto the shipped lm_head Marlin-W8 path. Qualified G0-G3 + counterbalanced A/B (-5.45 ms/frame end-to-end, 78.7 ms/frame, realtime structural gate ACCEPT); provenance and derivation tooling in the repo's deploy/promotion-candidate/. nano composite dd4ce6a99d03847c8da10708a0214b1815df078a5c43cae9846b7fe136085ce6.
- manifests/release.json +21 -16
- nano/artifact.sha256.json +14 -0
- nano/config.json +5 -4
- nano/model-00002-of-00004.safetensors +2 -2
- nano/model-00004-of-00004.safetensors +2 -2
- nano/model.safetensors.index.json +5 -2
- nano/quantize_config.json +3 -3
manifests/release.json
CHANGED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
{
|
| 2 |
-
"candidate": "production-candidate-
|
| 3 |
"community_release": true,
|
| 4 |
"corpus_release_sha256": "ad8ce2fe3a66fae72ce45703216265180066b2745dc32d829ee550b45c487bd5",
|
| 5 |
"derived_from_ea": false,
|
|
@@ -11,9 +11,9 @@
|
|
| 11 |
"sha256": "c55ed9a3dd7c5df14a2496d8ff0f6b1941f807af1769b09611012c4ec56960a0"
|
| 12 |
},
|
| 13 |
{
|
| 14 |
-
"bytes":
|
| 15 |
"path": "README.md",
|
| 16 |
-
"sha256": "
|
| 17 |
},
|
| 18 |
{
|
| 19 |
"bytes": 2141,
|
|
@@ -95,15 +95,20 @@
|
|
| 95 |
"path": "eartts/quantize_config.json",
|
| 96 |
"sha256": "2bbd43e9c7715806ca2b2d4dbebd58b04dbb74d1c07e29397038e056b4ec409b"
|
| 97 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 98 |
{
|
| 99 |
"bytes": 4070,
|
| 100 |
"path": "nano/chat_template.jinja",
|
| 101 |
"sha256": "b7a3a520a4bc1beae6a25d615f9215580df8eea926f488256c2bef74039433c0"
|
| 102 |
},
|
| 103 |
{
|
| 104 |
-
"bytes":
|
| 105 |
"path": "nano/config.json",
|
| 106 |
-
"sha256": "
|
| 107 |
},
|
| 108 |
{
|
| 109 |
"bytes": 12176,
|
|
@@ -116,9 +121,9 @@
|
|
| 116 |
"sha256": "ff1f20fb81c18448e8744cc8a6142ee817a6a4d47b99571b370d58e7d86f92a8"
|
| 117 |
},
|
| 118 |
{
|
| 119 |
-
"bytes":
|
| 120 |
"path": "nano/model-00002-of-00004.safetensors",
|
| 121 |
-
"sha256": "
|
| 122 |
},
|
| 123 |
{
|
| 124 |
"bytes": 4259170776,
|
|
@@ -126,19 +131,19 @@
|
|
| 126 |
"sha256": "8def47e0827bc1678804afce6a003f591bca58bbc2cc7ef11c01d8b802aebf21"
|
| 127 |
},
|
| 128 |
{
|
| 129 |
-
"bytes":
|
| 130 |
"path": "nano/model-00004-of-00004.safetensors",
|
| 131 |
-
"sha256": "
|
| 132 |
},
|
| 133 |
{
|
| 134 |
-
"bytes":
|
| 135 |
"path": "nano/model.safetensors.index.json",
|
| 136 |
-
"sha256": "
|
| 137 |
},
|
| 138 |
{
|
| 139 |
-
"bytes":
|
| 140 |
"path": "nano/quantize_config.json",
|
| 141 |
-
"sha256": "
|
| 142 |
},
|
| 143 |
{
|
| 144 |
"bytes": 422,
|
|
@@ -166,9 +171,9 @@
|
|
| 166 |
"sha256": "2348f90db187fdf1234f9b56ee4917983b84c1b5dc3942dd2522c36b43883f3a"
|
| 167 |
}
|
| 168 |
],
|
| 169 |
-
"files_sha256": "
|
| 170 |
"kind": "nemotron_voicechat_dgx_spark_hf_release",
|
| 171 |
-
"nano_composite_sha256": "
|
| 172 |
"nvidia_release": false,
|
| 173 |
"parent": {
|
| 174 |
"license": "OpenMDW-1.1",
|
|
@@ -176,7 +181,7 @@
|
|
| 176 |
"repository": "nvidia/NVIDIA-NemotronLabs-VoiceChat-11B",
|
| 177 |
"revision": "fb0f94eaf4d03ddc430f39565229393fa1b50c26"
|
| 178 |
},
|
| 179 |
-
"release_sha256": "
|
| 180 |
"repository": "pipecat-ai/NVIDIA-NemotronLabs-VoiceChat-11B-Spark",
|
| 181 |
"runtime_contract": {
|
| 182 |
"agent_no_audio_frames": 30,
|
|
|
|
| 1 |
{
|
| 2 |
+
"candidate": "production-candidate-3-fhw8",
|
| 3 |
"community_release": true,
|
| 4 |
"corpus_release_sha256": "ad8ce2fe3a66fae72ce45703216265180066b2745dc32d829ee550b45c487bd5",
|
| 5 |
"derived_from_ea": false,
|
|
|
|
| 11 |
"sha256": "c55ed9a3dd7c5df14a2496d8ff0f6b1941f807af1769b09611012c4ec56960a0"
|
| 12 |
},
|
| 13 |
{
|
| 14 |
+
"bytes": 4022,
|
| 15 |
"path": "README.md",
|
| 16 |
+
"sha256": "61c13082845fc40f25485f0367c2cfcc1b6726d3ad167420ed98d1c067f96e14"
|
| 17 |
},
|
| 18 |
{
|
| 19 |
"bytes": 2141,
|
|
|
|
| 95 |
"path": "eartts/quantize_config.json",
|
| 96 |
"sha256": "2bbd43e9c7715806ca2b2d4dbebd58b04dbb74d1c07e29397038e056b4ec409b"
|
| 97 |
},
|
| 98 |
+
{
|
| 99 |
+
"bytes": 1182,
|
| 100 |
+
"path": "nano/artifact.sha256.json",
|
| 101 |
+
"sha256": "09ed428991123ebcd6ca300cc02452112c8b1eb3df2cdc080826ead39f1d3098"
|
| 102 |
+
},
|
| 103 |
{
|
| 104 |
"bytes": 4070,
|
| 105 |
"path": "nano/chat_template.jinja",
|
| 106 |
"sha256": "b7a3a520a4bc1beae6a25d615f9215580df8eea926f488256c2bef74039433c0"
|
| 107 |
},
|
| 108 |
{
|
| 109 |
+
"bytes": 3994,
|
| 110 |
"path": "nano/config.json",
|
| 111 |
+
"sha256": "289b47e1aa92bf31c7767ea3cdb934f148820ba087b3001418f2df3f3470f3d8"
|
| 112 |
},
|
| 113 |
{
|
| 114 |
"bytes": 12176,
|
|
|
|
| 121 |
"sha256": "ff1f20fb81c18448e8744cc8a6142ee817a6a4d47b99571b370d58e7d86f92a8"
|
| 122 |
},
|
| 123 |
{
|
| 124 |
+
"bytes": 1945452632,
|
| 125 |
"path": "nano/model-00002-of-00004.safetensors",
|
| 126 |
+
"sha256": "659f015df802e9ec6696def3a194b12610feaaa2d8fcbe6fb3ffa0490b49e32c"
|
| 127 |
},
|
| 128 |
{
|
| 129 |
"bytes": 4259170776,
|
|
|
|
| 131 |
"sha256": "8def47e0827bc1678804afce6a003f591bca58bbc2cc7ef11c01d8b802aebf21"
|
| 132 |
},
|
| 133 |
{
|
| 134 |
+
"bytes": 3564975568,
|
| 135 |
"path": "nano/model-00004-of-00004.safetensors",
|
| 136 |
+
"sha256": "ac118ef7f5262be37c37e44ff71b6f95598e0a2e9a8af2912e94aff627948f4a"
|
| 137 |
},
|
| 138 |
{
|
| 139 |
+
"bytes": 56678,
|
| 140 |
"path": "nano/model.safetensors.index.json",
|
| 141 |
+
"sha256": "70fd0aa041ea5c4c121032b676c5fbe580f66ebce8b5fa948101ff58be2c2e0f"
|
| 142 |
},
|
| 143 |
{
|
| 144 |
+
"bytes": 429,
|
| 145 |
"path": "nano/quantize_config.json",
|
| 146 |
+
"sha256": "32008b58f9497fc019c994ee8b49af6751461b989b696d497ee807fa7f161354"
|
| 147 |
},
|
| 148 |
{
|
| 149 |
"bytes": 422,
|
|
|
|
| 171 |
"sha256": "2348f90db187fdf1234f9b56ee4917983b84c1b5dc3942dd2522c36b43883f3a"
|
| 172 |
}
|
| 173 |
],
|
| 174 |
+
"files_sha256": "81871ce73e6a42125275fbc0cffa8156c64863c64540042616f6594e38800411",
|
| 175 |
"kind": "nemotron_voicechat_dgx_spark_hf_release",
|
| 176 |
+
"nano_composite_sha256": "dd4ce6a99d03847c8da10708a0214b1815df078a5c43cae9846b7fe136085ce6",
|
| 177 |
"nvidia_release": false,
|
| 178 |
"parent": {
|
| 179 |
"license": "OpenMDW-1.1",
|
|
|
|
| 181 |
"repository": "nvidia/NVIDIA-NemotronLabs-VoiceChat-11B",
|
| 182 |
"revision": "fb0f94eaf4d03ddc430f39565229393fa1b50c26"
|
| 183 |
},
|
| 184 |
+
"release_sha256": "426eecc07cffd86c62d78f8498051dd8e5d3aa869f766535682a39bd5f7bd947",
|
| 185 |
"repository": "pipecat-ai/NVIDIA-NemotronLabs-VoiceChat-11B-Spark",
|
| 186 |
"runtime_contract": {
|
| 187 |
"agent_no_audio_frames": 30,
|
nano/artifact.sha256.json
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chat_template.jinja": "b7a3a520a4bc1beae6a25d615f9215580df8eea926f488256c2bef74039433c0",
|
| 3 |
+
"config.json": "289b47e1aa92bf31c7767ea3cdb934f148820ba087b3001418f2df3f3470f3d8",
|
| 4 |
+
"configuration_nemotron_h.py": "18186204622431c336a7c2813a80f0d37937dde962b8aebddf24caf5bffa03f7",
|
| 5 |
+
"model-00001-of-00004.safetensors": "ff1f20fb81c18448e8744cc8a6142ee817a6a4d47b99571b370d58e7d86f92a8",
|
| 6 |
+
"model-00002-of-00004.safetensors": "659f015df802e9ec6696def3a194b12610feaaa2d8fcbe6fb3ffa0490b49e32c",
|
| 7 |
+
"model-00003-of-00004.safetensors": "8def47e0827bc1678804afce6a003f591bca58bbc2cc7ef11c01d8b802aebf21",
|
| 8 |
+
"model-00004-of-00004.safetensors": "ac118ef7f5262be37c37e44ff71b6f95598e0a2e9a8af2912e94aff627948f4a",
|
| 9 |
+
"model.safetensors.index.json": "70fd0aa041ea5c4c121032b676c5fbe580f66ebce8b5fa948101ff58be2c2e0f",
|
| 10 |
+
"quantize_config.json": "32008b58f9497fc019c994ee8b49af6751461b989b696d497ee807fa7f161354",
|
| 11 |
+
"special_tokens_map.json": "2a4d2e7403546286e5d75f5b6b3c197490be67fb1e2118e5c60ad5c26e6668b1",
|
| 12 |
+
"tokenizer.json": "3277c00fe5fb3963b3cb7c07b7f183722d2af4d775a4aea7cfb3684d7cccbc2f",
|
| 13 |
+
"tokenizer_config.json": "d05c050a1cffa3b00df7b0d2596425a17a3e5a03159ea04e0457f6cdf8d29cd4"
|
| 14 |
+
}
|
nano/config.json
CHANGED
|
@@ -56,8 +56,7 @@
|
|
| 56 |
"desc_act": false,
|
| 57 |
"dynamic": {
|
| 58 |
"-:.*(?:embed_tokens|embed_asr_tokens)$": {},
|
| 59 |
-
"-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}
|
| 60 |
-
"-:.*function_head$": {}
|
| 61 |
},
|
| 62 |
"group_size": 128,
|
| 63 |
"lm_head": true,
|
|
@@ -66,7 +65,8 @@
|
|
| 66 |
"mixer.down_proj",
|
| 67 |
"mixer.in_proj",
|
| 68 |
"mixer.out_proj",
|
| 69 |
-
"lm_head"
|
|
|
|
| 70 |
],
|
| 71 |
"quant_method": "gptq",
|
| 72 |
"sym": true
|
|
@@ -125,7 +125,8 @@
|
|
| 125 |
},
|
| 126 |
"voicechat_pad_pair_token_id": 12,
|
| 127 |
"voicechat_quantized_output_heads": [
|
| 128 |
-
"lm_head"
|
|
|
|
| 129 |
],
|
| 130 |
"voicechat_skip_custom_text_logits": {
|
| 131 |
"contract": "temperature=0, top_p=1, repetition_penalty=1",
|
|
|
|
| 56 |
"desc_act": false,
|
| 57 |
"dynamic": {
|
| 58 |
"-:.*(?:embed_tokens|embed_asr_tokens)$": {},
|
| 59 |
+
"-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}
|
|
|
|
| 60 |
},
|
| 61 |
"group_size": 128,
|
| 62 |
"lm_head": true,
|
|
|
|
| 65 |
"mixer.down_proj",
|
| 66 |
"mixer.in_proj",
|
| 67 |
"mixer.out_proj",
|
| 68 |
+
"lm_head",
|
| 69 |
+
"function_head"
|
| 70 |
],
|
| 71 |
"quant_method": "gptq",
|
| 72 |
"sym": true
|
|
|
|
| 125 |
},
|
| 126 |
"voicechat_pad_pair_token_id": 12,
|
| 127 |
"voicechat_quantized_output_heads": [
|
| 128 |
+
"lm_head",
|
| 129 |
+
"function_head"
|
| 130 |
],
|
| 131 |
"voicechat_skip_custom_text_logits": {
|
| 132 |
"contract": "temperature=0, top_p=1, repetition_penalty=1",
|
nano/model-00002-of-00004.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:659f015df802e9ec6696def3a194b12610feaaa2d8fcbe6fb3ffa0490b49e32c
|
| 3 |
+
size 1945452632
|
nano/model-00004-of-00004.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ac118ef7f5262be37c37e44ff71b6f95598e0a2e9a8af2912e94aff627948f4a
|
| 3 |
+
size 3564975568
|
nano/model.safetensors.index.json
CHANGED
|
@@ -1,10 +1,13 @@
|
|
| 1 |
{
|
| 2 |
"metadata": {
|
| 3 |
-
"total_size":
|
| 4 |
},
|
| 5 |
"weight_map": {
|
| 6 |
"stt_model.embed_tokens.weight": "model-00001-of-00004.safetensors",
|
| 7 |
-
"stt_model.function_head.
|
|
|
|
|
|
|
|
|
|
| 8 |
"stt_model.llm.layers.0.mixer.A_log": "model-00002-of-00004.safetensors",
|
| 9 |
"stt_model.llm.layers.0.mixer.D": "model-00002-of-00004.safetensors",
|
| 10 |
"stt_model.llm.layers.0.mixer.conv1d.bias": "model-00002-of-00004.safetensors",
|
|
|
|
| 1 |
{
|
| 2 |
"metadata": {
|
| 3 |
+
"total_size": 12118332992
|
| 4 |
},
|
| 5 |
"weight_map": {
|
| 6 |
"stt_model.embed_tokens.weight": "model-00001-of-00004.safetensors",
|
| 7 |
+
"stt_model.function_head.g_idx": "model-00004-of-00004.safetensors",
|
| 8 |
+
"stt_model.function_head.qweight": "model-00004-of-00004.safetensors",
|
| 9 |
+
"stt_model.function_head.qzeros": "model-00004-of-00004.safetensors",
|
| 10 |
+
"stt_model.function_head.scales": "model-00004-of-00004.safetensors",
|
| 11 |
"stt_model.llm.layers.0.mixer.A_log": "model-00002-of-00004.safetensors",
|
| 12 |
"stt_model.llm.layers.0.mixer.D": "model-00002-of-00004.safetensors",
|
| 13 |
"stt_model.llm.layers.0.mixer.conv1d.bias": "model-00002-of-00004.safetensors",
|
nano/quantize_config.json
CHANGED
|
@@ -4,8 +4,7 @@
|
|
| 4 |
"desc_act": false,
|
| 5 |
"dynamic": {
|
| 6 |
"-:.*(?:embed_tokens|embed_asr_tokens)$": {},
|
| 7 |
-
"-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}
|
| 8 |
-
"-:.*function_head$": {}
|
| 9 |
},
|
| 10 |
"group_size": 128,
|
| 11 |
"lm_head": true,
|
|
@@ -14,7 +13,8 @@
|
|
| 14 |
"mixer.down_proj",
|
| 15 |
"mixer.in_proj",
|
| 16 |
"mixer.out_proj",
|
| 17 |
-
"lm_head"
|
|
|
|
| 18 |
],
|
| 19 |
"quant_method": "gptq",
|
| 20 |
"sym": true
|
|
|
|
| 4 |
"desc_act": false,
|
| 5 |
"dynamic": {
|
| 6 |
"-:.*(?:embed_tokens|embed_asr_tokens)$": {},
|
| 7 |
+
"-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}
|
|
|
|
| 8 |
},
|
| 9 |
"group_size": 128,
|
| 10 |
"lm_head": true,
|
|
|
|
| 13 |
"mixer.down_proj",
|
| 14 |
"mixer.in_proj",
|
| 15 |
"mixer.out_proj",
|
| 16 |
+
"lm_head",
|
| 17 |
+
"function_head"
|
| 18 |
],
|
| 19 |
"quant_method": "gptq",
|
| 20 |
"sym": true
|