Kwindla commited on
Commit
fdf70ed
·
verified ·
1 Parent(s): 547602d

fhw8: quantize function_head to GPTQ-W8 (production-candidate-3-fhw8)

Browse files

Deterministic RTN requantization of the F32 function_head onto the shipped lm_head Marlin-W8 path. Qualified G0-G3 + counterbalanced A/B (-5.45 ms/frame end-to-end, 78.7 ms/frame, realtime structural gate ACCEPT); provenance and derivation tooling in the repo's deploy/promotion-candidate/. nano composite dd4ce6a99d03847c8da10708a0214b1815df078a5c43cae9846b7fe136085ce6.

manifests/release.json CHANGED
@@ -1,5 +1,5 @@
1
  {
2
- "candidate": "production-candidate-1",
3
  "community_release": true,
4
  "corpus_release_sha256": "ad8ce2fe3a66fae72ce45703216265180066b2745dc32d829ee550b45c487bd5",
5
  "derived_from_ea": false,
@@ -11,9 +11,9 @@
11
  "sha256": "c55ed9a3dd7c5df14a2496d8ff0f6b1941f807af1769b09611012c4ec56960a0"
12
  },
13
  {
14
- "bytes": 4496,
15
  "path": "README.md",
16
- "sha256": "36d3ec75c35406ca5105898c5c516883c92564e3f60bc1a3d227501e9e8cc5e4"
17
  },
18
  {
19
  "bytes": 2141,
@@ -95,15 +95,20 @@
95
  "path": "eartts/quantize_config.json",
96
  "sha256": "2bbd43e9c7715806ca2b2d4dbebd58b04dbb74d1c07e29397038e056b4ec409b"
97
  },
 
 
 
 
 
98
  {
99
  "bytes": 4070,
100
  "path": "nano/chat_template.jinja",
101
  "sha256": "b7a3a520a4bc1beae6a25d615f9215580df8eea926f488256c2bef74039433c0"
102
  },
103
  {
104
- "bytes": 3982,
105
  "path": "nano/config.json",
106
- "sha256": "2a0576a8b71aa09a5a6683d5a0d2e3fc099839ea9fe559d69a3458e828530391"
107
  },
108
  {
109
  "bytes": 12176,
@@ -116,9 +121,9 @@
116
  "sha256": "ff1f20fb81c18448e8744cc8a6142ee817a6a4d47b99571b370d58e7d86f92a8"
117
  },
118
  {
119
- "bytes": 4294263336,
120
  "path": "nano/model-00002-of-00004.safetensors",
121
- "sha256": "38ff5efa8d7de903d0dd45ed5cf22212db3cbc5ab87e99a3f2f010ac4ca63216"
122
  },
123
  {
124
  "bytes": 4259170776,
@@ -126,19 +131,19 @@
126
  "sha256": "8def47e0827bc1678804afce6a003f591bca58bbc2cc7ef11c01d8b802aebf21"
127
  },
128
  {
129
- "bytes": 2963992080,
130
  "path": "nano/model-00004-of-00004.safetensors",
131
- "sha256": "6d0f28a5cfa176c82e673e09f113b7e03e3f8ad4cb491f33b30501c7f767fdea"
132
  },
133
  {
134
- "bytes": 56456,
135
  "path": "nano/model.safetensors.index.json",
136
- "sha256": "a83fcd79b1661af926ddf62eac663a269853443c3d03dc690da6928598a9fbad"
137
  },
138
  {
139
- "bytes": 438,
140
  "path": "nano/quantize_config.json",
141
- "sha256": "7eeba0b4d95e514ec6ba40a4c917bdde900282ce512eee1b509467078b5a58ec"
142
  },
143
  {
144
  "bytes": 422,
@@ -166,9 +171,9 @@
166
  "sha256": "2348f90db187fdf1234f9b56ee4917983b84c1b5dc3942dd2522c36b43883f3a"
167
  }
168
  ],
169
- "files_sha256": "1665273ad23e289bcf9f7b7caf14813c97532c541721ad89682b87bf13a2ae1a",
170
  "kind": "nemotron_voicechat_dgx_spark_hf_release",
171
- "nano_composite_sha256": "3acecd057f7367768d621bec649cbe6cbe327ca0857c17f86f7e0929f2bb2d99",
172
  "nvidia_release": false,
173
  "parent": {
174
  "license": "OpenMDW-1.1",
@@ -176,7 +181,7 @@
176
  "repository": "nvidia/NVIDIA-NemotronLabs-VoiceChat-11B",
177
  "revision": "fb0f94eaf4d03ddc430f39565229393fa1b50c26"
178
  },
179
- "release_sha256": "ab2a14265d0ffe2751d02f8a628c547ad95c8d6a8b7c202470b6b8597dd86e5e",
180
  "repository": "pipecat-ai/NVIDIA-NemotronLabs-VoiceChat-11B-Spark",
181
  "runtime_contract": {
182
  "agent_no_audio_frames": 30,
 
1
  {
2
+ "candidate": "production-candidate-3-fhw8",
3
  "community_release": true,
4
  "corpus_release_sha256": "ad8ce2fe3a66fae72ce45703216265180066b2745dc32d829ee550b45c487bd5",
5
  "derived_from_ea": false,
 
11
  "sha256": "c55ed9a3dd7c5df14a2496d8ff0f6b1941f807af1769b09611012c4ec56960a0"
12
  },
13
  {
14
+ "bytes": 4022,
15
  "path": "README.md",
16
+ "sha256": "61c13082845fc40f25485f0367c2cfcc1b6726d3ad167420ed98d1c067f96e14"
17
  },
18
  {
19
  "bytes": 2141,
 
95
  "path": "eartts/quantize_config.json",
96
  "sha256": "2bbd43e9c7715806ca2b2d4dbebd58b04dbb74d1c07e29397038e056b4ec409b"
97
  },
98
+ {
99
+ "bytes": 1182,
100
+ "path": "nano/artifact.sha256.json",
101
+ "sha256": "09ed428991123ebcd6ca300cc02452112c8b1eb3df2cdc080826ead39f1d3098"
102
+ },
103
  {
104
  "bytes": 4070,
105
  "path": "nano/chat_template.jinja",
106
  "sha256": "b7a3a520a4bc1beae6a25d615f9215580df8eea926f488256c2bef74039433c0"
107
  },
108
  {
109
+ "bytes": 3994,
110
  "path": "nano/config.json",
111
+ "sha256": "289b47e1aa92bf31c7767ea3cdb934f148820ba087b3001418f2df3f3470f3d8"
112
  },
113
  {
114
  "bytes": 12176,
 
121
  "sha256": "ff1f20fb81c18448e8744cc8a6142ee817a6a4d47b99571b370d58e7d86f92a8"
122
  },
123
  {
124
+ "bytes": 1945452632,
125
  "path": "nano/model-00002-of-00004.safetensors",
126
+ "sha256": "659f015df802e9ec6696def3a194b12610feaaa2d8fcbe6fb3ffa0490b49e32c"
127
  },
128
  {
129
  "bytes": 4259170776,
 
131
  "sha256": "8def47e0827bc1678804afce6a003f591bca58bbc2cc7ef11c01d8b802aebf21"
132
  },
133
  {
134
+ "bytes": 3564975568,
135
  "path": "nano/model-00004-of-00004.safetensors",
136
+ "sha256": "ac118ef7f5262be37c37e44ff71b6f95598e0a2e9a8af2912e94aff627948f4a"
137
  },
138
  {
139
+ "bytes": 56678,
140
  "path": "nano/model.safetensors.index.json",
141
+ "sha256": "70fd0aa041ea5c4c121032b676c5fbe580f66ebce8b5fa948101ff58be2c2e0f"
142
  },
143
  {
144
+ "bytes": 429,
145
  "path": "nano/quantize_config.json",
146
+ "sha256": "32008b58f9497fc019c994ee8b49af6751461b989b696d497ee807fa7f161354"
147
  },
148
  {
149
  "bytes": 422,
 
171
  "sha256": "2348f90db187fdf1234f9b56ee4917983b84c1b5dc3942dd2522c36b43883f3a"
172
  }
173
  ],
174
+ "files_sha256": "81871ce73e6a42125275fbc0cffa8156c64863c64540042616f6594e38800411",
175
  "kind": "nemotron_voicechat_dgx_spark_hf_release",
176
+ "nano_composite_sha256": "dd4ce6a99d03847c8da10708a0214b1815df078a5c43cae9846b7fe136085ce6",
177
  "nvidia_release": false,
178
  "parent": {
179
  "license": "OpenMDW-1.1",
 
181
  "repository": "nvidia/NVIDIA-NemotronLabs-VoiceChat-11B",
182
  "revision": "fb0f94eaf4d03ddc430f39565229393fa1b50c26"
183
  },
184
+ "release_sha256": "426eecc07cffd86c62d78f8498051dd8e5d3aa869f766535682a39bd5f7bd947",
185
  "repository": "pipecat-ai/NVIDIA-NemotronLabs-VoiceChat-11B-Spark",
186
  "runtime_contract": {
187
  "agent_no_audio_frames": 30,
nano/artifact.sha256.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chat_template.jinja": "b7a3a520a4bc1beae6a25d615f9215580df8eea926f488256c2bef74039433c0",
3
+ "config.json": "289b47e1aa92bf31c7767ea3cdb934f148820ba087b3001418f2df3f3470f3d8",
4
+ "configuration_nemotron_h.py": "18186204622431c336a7c2813a80f0d37937dde962b8aebddf24caf5bffa03f7",
5
+ "model-00001-of-00004.safetensors": "ff1f20fb81c18448e8744cc8a6142ee817a6a4d47b99571b370d58e7d86f92a8",
6
+ "model-00002-of-00004.safetensors": "659f015df802e9ec6696def3a194b12610feaaa2d8fcbe6fb3ffa0490b49e32c",
7
+ "model-00003-of-00004.safetensors": "8def47e0827bc1678804afce6a003f591bca58bbc2cc7ef11c01d8b802aebf21",
8
+ "model-00004-of-00004.safetensors": "ac118ef7f5262be37c37e44ff71b6f95598e0a2e9a8af2912e94aff627948f4a",
9
+ "model.safetensors.index.json": "70fd0aa041ea5c4c121032b676c5fbe580f66ebce8b5fa948101ff58be2c2e0f",
10
+ "quantize_config.json": "32008b58f9497fc019c994ee8b49af6751461b989b696d497ee807fa7f161354",
11
+ "special_tokens_map.json": "2a4d2e7403546286e5d75f5b6b3c197490be67fb1e2118e5c60ad5c26e6668b1",
12
+ "tokenizer.json": "3277c00fe5fb3963b3cb7c07b7f183722d2af4d775a4aea7cfb3684d7cccbc2f",
13
+ "tokenizer_config.json": "d05c050a1cffa3b00df7b0d2596425a17a3e5a03159ea04e0457f6cdf8d29cd4"
14
+ }
nano/config.json CHANGED
@@ -56,8 +56,7 @@
56
  "desc_act": false,
57
  "dynamic": {
58
  "-:.*(?:embed_tokens|embed_asr_tokens)$": {},
59
- "-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {},
60
- "-:.*function_head$": {}
61
  },
62
  "group_size": 128,
63
  "lm_head": true,
@@ -66,7 +65,8 @@
66
  "mixer.down_proj",
67
  "mixer.in_proj",
68
  "mixer.out_proj",
69
- "lm_head"
 
70
  ],
71
  "quant_method": "gptq",
72
  "sym": true
@@ -125,7 +125,8 @@
125
  },
126
  "voicechat_pad_pair_token_id": 12,
127
  "voicechat_quantized_output_heads": [
128
- "lm_head"
 
129
  ],
130
  "voicechat_skip_custom_text_logits": {
131
  "contract": "temperature=0, top_p=1, repetition_penalty=1",
 
56
  "desc_act": false,
57
  "dynamic": {
58
  "-:.*(?:embed_tokens|embed_asr_tokens)$": {},
59
+ "-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}
 
60
  },
61
  "group_size": 128,
62
  "lm_head": true,
 
65
  "mixer.down_proj",
66
  "mixer.in_proj",
67
  "mixer.out_proj",
68
+ "lm_head",
69
+ "function_head"
70
  ],
71
  "quant_method": "gptq",
72
  "sym": true
 
125
  },
126
  "voicechat_pad_pair_token_id": 12,
127
  "voicechat_quantized_output_heads": [
128
+ "lm_head",
129
+ "function_head"
130
  ],
131
  "voicechat_skip_custom_text_logits": {
132
  "contract": "temperature=0, top_p=1, repetition_penalty=1",
nano/model-00002-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:38ff5efa8d7de903d0dd45ed5cf22212db3cbc5ab87e99a3f2f010ac4ca63216
3
- size 4294263336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:659f015df802e9ec6696def3a194b12610feaaa2d8fcbe6fb3ffa0490b49e32c
3
+ size 1945452632
nano/model-00004-of-00004.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6d0f28a5cfa176c82e673e09f113b7e03e3f8ad4cb491f33b30501c7f767fdea
3
- size 2963992080
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac118ef7f5262be37c37e44ff71b6f95598e0a2e9a8af2912e94aff627948f4a
3
+ size 3564975568
nano/model.safetensors.index.json CHANGED
@@ -1,10 +1,13 @@
1
  {
2
  "metadata": {
3
- "total_size": 13866160192
4
  },
5
  "weight_map": {
6
  "stt_model.embed_tokens.weight": "model-00001-of-00004.safetensors",
7
- "stt_model.function_head.weight": "model-00002-of-00004.safetensors",
 
 
 
8
  "stt_model.llm.layers.0.mixer.A_log": "model-00002-of-00004.safetensors",
9
  "stt_model.llm.layers.0.mixer.D": "model-00002-of-00004.safetensors",
10
  "stt_model.llm.layers.0.mixer.conv1d.bias": "model-00002-of-00004.safetensors",
 
1
  {
2
  "metadata": {
3
+ "total_size": 12118332992
4
  },
5
  "weight_map": {
6
  "stt_model.embed_tokens.weight": "model-00001-of-00004.safetensors",
7
+ "stt_model.function_head.g_idx": "model-00004-of-00004.safetensors",
8
+ "stt_model.function_head.qweight": "model-00004-of-00004.safetensors",
9
+ "stt_model.function_head.qzeros": "model-00004-of-00004.safetensors",
10
+ "stt_model.function_head.scales": "model-00004-of-00004.safetensors",
11
  "stt_model.llm.layers.0.mixer.A_log": "model-00002-of-00004.safetensors",
12
  "stt_model.llm.layers.0.mixer.D": "model-00002-of-00004.safetensors",
13
  "stt_model.llm.layers.0.mixer.conv1d.bias": "model-00002-of-00004.safetensors",
nano/quantize_config.json CHANGED
@@ -4,8 +4,7 @@
4
  "desc_act": false,
5
  "dynamic": {
6
  "-:.*(?:embed_tokens|embed_asr_tokens)$": {},
7
- "-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {},
8
- "-:.*function_head$": {}
9
  },
10
  "group_size": 128,
11
  "lm_head": true,
@@ -14,7 +13,8 @@
14
  "mixer.down_proj",
15
  "mixer.in_proj",
16
  "mixer.out_proj",
17
- "lm_head"
 
18
  ],
19
  "quant_method": "gptq",
20
  "sym": true
 
4
  "desc_act": false,
5
  "dynamic": {
6
  "-:.*(?:embed_tokens|embed_asr_tokens)$": {},
7
+ "-:.*(?:q_proj|k_proj|v_proj|o_proj)$": {}
 
8
  },
9
  "group_size": 128,
10
  "lm_head": true,
 
13
  "mixer.down_proj",
14
  "mixer.in_proj",
15
  "mixer.out_proj",
16
+ "lm_head",
17
+ "function_head"
18
  ],
19
  "quant_method": "gptq",
20
  "sym": true