v3.1 matrices, three-seed benchmark and eleven-language demo (part 2)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +236 -0
- README.md +173 -192
- bench3.1/README.md +494 -0
- bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/ru/4b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/ru/4b-v3.1_s100000.flac +3 -0
- bench3.1/audio/ru/4b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/ru/4b-v3.1_s42.flac +3 -0
- bench3.1/audio/ru/4b-v3_s100000.flac +3 -0
- bench3.1/audio/ru/4b-v3_s100000000.flac +3 -0
- bench3.1/audio/ru/4b-v3_s42.flac +3 -0
- bench3.1/audio/ru/8b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/ru/8b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/ru/8b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac +3 -0
- bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/ru/8b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/ru/8b-v3.1_s100000.flac +3 -0
- bench3.1/audio/ru/8b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/ru/8b-v3.1_s42.flac +3 -0
- bench3.1/audio/ru/8b-v3_s100000.flac +3 -0
- bench3.1/audio/ru/8b-v3_s100000000.flac +3 -0
- bench3.1/audio/ru/8b-v3_s42.flac +3 -0
- bench3.1/audio/zh/32b_s100000.flac +3 -0
- bench3.1/audio/zh/32b_s100000000.flac +3 -0
- bench3.1/audio/zh/32b_s42.flac +3 -0
- bench3.1/audio/zh/4b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/zh/4b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/zh/4b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac +3 -0
- bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/zh/4b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/zh/4b-v3.1_s100000.flac +3 -0
- bench3.1/audio/zh/4b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/zh/4b-v3.1_s42.flac +3 -0
- bench3.1/audio/zh/4b-v3_s100000.flac +3 -0
- bench3.1/audio/zh/4b-v3_s100000000.flac +3 -0
- bench3.1/audio/zh/4b-v3_s42.flac +3 -0
- bench3.1/audio/zh/8b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/zh/8b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/zh/8b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac +3 -0
- bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/zh/8b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/zh/8b-v3.1_s100000.flac +3 -0
- bench3.1/audio/zh/8b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/zh/8b-v3.1_s42.flac +3 -0
- bench3.1/audio/zh/8b-v3_s100000.flac +3 -0
- bench3.1/audio/zh/8b-v3_s100000000.flac +3 -0
- bench3.1/audio/zh/8b-v3_s42.flac +3 -0
.gitattributes
CHANGED
|
@@ -289,3 +289,239 @@ bench3.1/audio/ru/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
|
| 289 |
bench3.1/audio/ru/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 290 |
bench3.1/audio/ru/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 291 |
bench3.1/audio/ru/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 289 |
bench3.1/audio/ru/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 290 |
bench3.1/audio/ru/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 291 |
bench3.1/audio/ru/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 292 |
+
bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 293 |
+
bench3.1/audio/ru/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 294 |
+
bench3.1/audio/ru/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 295 |
+
bench3.1/audio/ru/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 296 |
+
bench3.1/audio/ru/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 297 |
+
bench3.1/audio/ru/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 298 |
+
bench3.1/audio/ru/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 299 |
+
bench3.1/audio/ru/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 300 |
+
bench3.1/audio/ru/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 301 |
+
bench3.1/audio/ru/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 302 |
+
bench3.1/audio/ru/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 303 |
+
bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 304 |
+
bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 305 |
+
bench3.1/audio/ru/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 306 |
+
bench3.1/audio/ru/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 307 |
+
bench3.1/audio/ru/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 308 |
+
bench3.1/audio/ru/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 309 |
+
bench3.1/audio/ru/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 310 |
+
bench3.1/audio/ru/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 311 |
+
bench3.1/audio/ru/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 312 |
+
bench3.1/audio/zh/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 313 |
+
bench3.1/audio/zh/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 314 |
+
bench3.1/audio/zh/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 315 |
+
bench3.1/audio/zh/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 316 |
+
bench3.1/audio/zh/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 317 |
+
bench3.1/audio/zh/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 318 |
+
bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 319 |
+
bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 320 |
+
bench3.1/audio/zh/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 321 |
+
bench3.1/audio/zh/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 322 |
+
bench3.1/audio/zh/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 323 |
+
bench3.1/audio/zh/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 324 |
+
bench3.1/audio/zh/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 325 |
+
bench3.1/audio/zh/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 326 |
+
bench3.1/audio/zh/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 327 |
+
bench3.1/audio/zh/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 328 |
+
bench3.1/audio/zh/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 329 |
+
bench3.1/audio/zh/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 330 |
+
bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 331 |
+
bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 332 |
+
bench3.1/audio/zh/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 333 |
+
bench3.1/audio/zh/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 334 |
+
bench3.1/audio/zh/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 335 |
+
bench3.1/audio/zh/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 336 |
+
bench3.1/audio/zh/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 337 |
+
bench3.1/audio/zh/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 338 |
+
bench3.1/audio/zh/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 339 |
+
bench3.1/images/brutes/p01_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 340 |
+
bench3.1/images/brutes/p01_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 341 |
+
bench3.1/images/brutes/p01_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 342 |
+
bench3.1/images/brutes/p01_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 343 |
+
bench3.1/images/brutes/p01_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 344 |
+
bench3.1/images/brutes/p02_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 345 |
+
bench3.1/images/brutes/p02_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 346 |
+
bench3.1/images/brutes/p02_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 347 |
+
bench3.1/images/brutes/p02_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 348 |
+
bench3.1/images/brutes/p02_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 349 |
+
bench3.1/images/brutes/p03_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 350 |
+
bench3.1/images/brutes/p03_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 351 |
+
bench3.1/images/brutes/p03_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 352 |
+
bench3.1/images/brutes/p03_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 353 |
+
bench3.1/images/brutes/p03_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 354 |
+
bench3.1/images/brutes/p04_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 355 |
+
bench3.1/images/brutes/p04_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 356 |
+
bench3.1/images/brutes/p04_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 357 |
+
bench3.1/images/brutes/p04_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 358 |
+
bench3.1/images/brutes/p04_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 359 |
+
bench3.1/images/brutes/p05_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 360 |
+
bench3.1/images/brutes/p05_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 361 |
+
bench3.1/images/brutes/p05_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 362 |
+
bench3.1/images/brutes/p05_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 363 |
+
bench3.1/images/brutes/p05_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 364 |
+
bench3.1/images/brutes/p06_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 365 |
+
bench3.1/images/brutes/p06_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 366 |
+
bench3.1/images/brutes/p06_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 367 |
+
bench3.1/images/brutes/p06_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 368 |
+
bench3.1/images/brutes/p06_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 369 |
+
bench3.1/images/brutes/p07_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 370 |
+
bench3.1/images/brutes/p07_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 371 |
+
bench3.1/images/brutes/p07_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 372 |
+
bench3.1/images/brutes/p07_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 373 |
+
bench3.1/images/brutes/p07_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 374 |
+
bench3.1/images/brutes/p08_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 375 |
+
bench3.1/images/brutes/p08_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 376 |
+
bench3.1/images/brutes/p08_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 377 |
+
bench3.1/images/brutes/p08_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 378 |
+
bench3.1/images/brutes/p08_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 379 |
+
bench3.1/images/brutes/p09_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 380 |
+
bench3.1/images/brutes/p09_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 381 |
+
bench3.1/images/brutes/p09_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 382 |
+
bench3.1/images/brutes/p09_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 383 |
+
bench3.1/images/brutes/p09_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 384 |
+
bench3.1/images/brutes/p10_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 385 |
+
bench3.1/images/brutes/p10_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 386 |
+
bench3.1/images/brutes/p10_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 387 |
+
bench3.1/images/brutes/p10_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 388 |
+
bench3.1/images/brutes/p10_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 389 |
+
bench3.1/images/brutes/p11_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 390 |
+
bench3.1/images/brutes/p11_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 391 |
+
bench3.1/images/brutes/p11_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 392 |
+
bench3.1/images/brutes/p11_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 393 |
+
bench3.1/images/brutes/p11_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 394 |
+
bench3.1/images/brutes/p12_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 395 |
+
bench3.1/images/brutes/p12_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 396 |
+
bench3.1/images/brutes/p12_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 397 |
+
bench3.1/images/brutes/p12_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 398 |
+
bench3.1/images/brutes/p12_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 399 |
+
bench3.1/images/brutes/p13_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 400 |
+
bench3.1/images/brutes/p13_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 401 |
+
bench3.1/images/brutes/p13_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 402 |
+
bench3.1/images/brutes/p13_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 403 |
+
bench3.1/images/brutes/p13_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 404 |
+
bench3.1/images/brutes/p14_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 405 |
+
bench3.1/images/brutes/p14_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 406 |
+
bench3.1/images/brutes/p14_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 407 |
+
bench3.1/images/brutes/p14_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 408 |
+
bench3.1/images/brutes/p14_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 409 |
+
bench3.1/images/brutes/p15_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 410 |
+
bench3.1/images/brutes/p15_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 411 |
+
bench3.1/images/brutes/p15_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 412 |
+
bench3.1/images/brutes/p15_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 413 |
+
bench3.1/images/brutes/p15_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 414 |
+
bench3.1/images/planches/p01.jpg filter=lfs diff=lfs merge=lfs -text
|
| 415 |
+
bench3.1/images/planches/p02.jpg filter=lfs diff=lfs merge=lfs -text
|
| 416 |
+
bench3.1/images/planches/p03.jpg filter=lfs diff=lfs merge=lfs -text
|
| 417 |
+
bench3.1/images/planches/p04.jpg filter=lfs diff=lfs merge=lfs -text
|
| 418 |
+
bench3.1/images/planches/p05.jpg filter=lfs diff=lfs merge=lfs -text
|
| 419 |
+
bench3.1/images/planches/p06.jpg filter=lfs diff=lfs merge=lfs -text
|
| 420 |
+
bench3.1/images/planches/p07.jpg filter=lfs diff=lfs merge=lfs -text
|
| 421 |
+
bench3.1/images/planches/p08.jpg filter=lfs diff=lfs merge=lfs -text
|
| 422 |
+
bench3.1/images/planches/p09.jpg filter=lfs diff=lfs merge=lfs -text
|
| 423 |
+
bench3.1/images/planches/p10.jpg filter=lfs diff=lfs merge=lfs -text
|
| 424 |
+
bench3.1/images/planches/p12.jpg filter=lfs diff=lfs merge=lfs -text
|
| 425 |
+
bench3.1/images/planches/p13.jpg filter=lfs diff=lfs merge=lfs -text
|
| 426 |
+
bench3.1/images/planches/p14.jpg filter=lfs diff=lfs merge=lfs -text
|
| 427 |
+
bench3.1/images/planches/p15.jpg filter=lfs diff=lfs merge=lfs -text
|
| 428 |
+
bench3.1/video/clipproj-v3.1-eleven-languages.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 429 |
+
bench3.1/video/langues/ar/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 430 |
+
bench3.1/video/langues/ar/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 431 |
+
bench3.1/video/langues/ar/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 432 |
+
bench3.1/video/langues/ar/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 433 |
+
bench3.1/video/langues/ar/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 434 |
+
bench3.1/video/langues/ar/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 435 |
+
bench3.1/video/langues/ar/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 436 |
+
bench3.1/video/langues/ar/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 437 |
+
bench3.1/video/langues/ar/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 438 |
+
bench3.1/video/langues/de/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 439 |
+
bench3.1/video/langues/de/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 440 |
+
bench3.1/video/langues/de/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 441 |
+
bench3.1/video/langues/de/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 442 |
+
bench3.1/video/langues/de/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 443 |
+
bench3.1/video/langues/de/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 444 |
+
bench3.1/video/langues/de/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 445 |
+
bench3.1/video/langues/de/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 446 |
+
bench3.1/video/langues/de/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 447 |
+
bench3.1/video/langues/en/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 448 |
+
bench3.1/video/langues/en/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 449 |
+
bench3.1/video/langues/en/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 450 |
+
bench3.1/video/langues/en/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 451 |
+
bench3.1/video/langues/en/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 452 |
+
bench3.1/video/langues/en/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 453 |
+
bench3.1/video/langues/en/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 454 |
+
bench3.1/video/langues/en/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 455 |
+
bench3.1/video/langues/en/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 456 |
+
bench3.1/video/langues/es/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 457 |
+
bench3.1/video/langues/es/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 458 |
+
bench3.1/video/langues/es/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 459 |
+
bench3.1/video/langues/es/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 460 |
+
bench3.1/video/langues/es/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 461 |
+
bench3.1/video/langues/es/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 462 |
+
bench3.1/video/langues/es/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 463 |
+
bench3.1/video/langues/es/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 464 |
+
bench3.1/video/langues/es/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 465 |
+
bench3.1/video/langues/fr/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 466 |
+
bench3.1/video/langues/fr/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 467 |
+
bench3.1/video/langues/fr/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 468 |
+
bench3.1/video/langues/fr/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 469 |
+
bench3.1/video/langues/fr/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 470 |
+
bench3.1/video/langues/fr/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 471 |
+
bench3.1/video/langues/fr/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 472 |
+
bench3.1/video/langues/fr/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 473 |
+
bench3.1/video/langues/fr/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 474 |
+
bench3.1/video/langues/it/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 475 |
+
bench3.1/video/langues/it/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 476 |
+
bench3.1/video/langues/it/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 477 |
+
bench3.1/video/langues/it/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 478 |
+
bench3.1/video/langues/it/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 479 |
+
bench3.1/video/langues/it/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 480 |
+
bench3.1/video/langues/it/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 481 |
+
bench3.1/video/langues/it/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 482 |
+
bench3.1/video/langues/it/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 483 |
+
bench3.1/video/langues/ja/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 484 |
+
bench3.1/video/langues/ja/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 485 |
+
bench3.1/video/langues/ja/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 486 |
+
bench3.1/video/langues/ja/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 487 |
+
bench3.1/video/langues/ja/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 488 |
+
bench3.1/video/langues/ja/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 489 |
+
bench3.1/video/langues/ja/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 490 |
+
bench3.1/video/langues/ja/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 491 |
+
bench3.1/video/langues/ja/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 492 |
+
bench3.1/video/langues/ko/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 493 |
+
bench3.1/video/langues/ko/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 494 |
+
bench3.1/video/langues/ko/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 495 |
+
bench3.1/video/langues/ko/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 496 |
+
bench3.1/video/langues/ko/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 497 |
+
bench3.1/video/langues/ko/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 498 |
+
bench3.1/video/langues/ko/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 499 |
+
bench3.1/video/langues/ko/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 500 |
+
bench3.1/video/langues/ko/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 501 |
+
bench3.1/video/langues/pt/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 502 |
+
bench3.1/video/langues/pt/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 503 |
+
bench3.1/video/langues/pt/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 504 |
+
bench3.1/video/langues/pt/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 505 |
+
bench3.1/video/langues/pt/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 506 |
+
bench3.1/video/langues/pt/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 507 |
+
bench3.1/video/langues/pt/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 508 |
+
bench3.1/video/langues/pt/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 509 |
+
bench3.1/video/langues/pt/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 510 |
+
bench3.1/video/langues/ru/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 511 |
+
bench3.1/video/langues/ru/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 512 |
+
bench3.1/video/langues/ru/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 513 |
+
bench3.1/video/langues/ru/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 514 |
+
bench3.1/video/langues/ru/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 515 |
+
bench3.1/video/langues/ru/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 516 |
+
bench3.1/video/langues/ru/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 517 |
+
bench3.1/video/langues/ru/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 518 |
+
bench3.1/video/langues/ru/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 519 |
+
bench3.1/video/langues/zh/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 520 |
+
bench3.1/video/langues/zh/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 521 |
+
bench3.1/video/langues/zh/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 522 |
+
bench3.1/video/langues/zh/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 523 |
+
bench3.1/video/langues/zh/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 524 |
+
bench3.1/video/langues/zh/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 525 |
+
bench3.1/video/langues/zh/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 526 |
+
bench3.1/video/langues/zh/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 527 |
+
bench3.1/video/langues/zh/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -6,226 +6,214 @@ tags:
|
|
| 6 |
- text-to-video
|
| 7 |
- qwen3-vl
|
| 8 |
- text-encoder
|
|
|
|
| 9 |
base_model:
|
| 10 |
- Comfy-Org/MiniMax-H3
|
| 11 |
- Qwen/Qwen3-VL-4B-Instruct
|
|
|
|
| 12 |
library_name: comfyui
|
| 13 |
---
|
| 14 |
|
| 15 |
# ClipProj — MiniMax H3 conditioning from a Qwen3-VL-4B or 8B
|
| 16 |
|
| 17 |
**Projection matrices that let a small Qwen3-VL replace the Qwen3-VL-32B text encoder of MiniMax H3.**
|
|
|
|
| 18 |
|
| 19 |
-
|
| 20 |
|
| 21 |
-
|
|
|
|
|
|
|
| 22 |
|
| 23 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
| Qwen3-VL-32B nvfp4 | 15.69 GB | — | **15.7 GB** |
|
| 28 |
-
| Qwen3-VL-8B int8 + `v3-mlp` | 10.01 GB | 604 MB | **10.6 GB** |
|
| 29 |
-
| Qwen3-VL-8B int8 + `v3` | 10.01 GB | 84 MB | **10.1 GB** |
|
| 30 |
-
| Qwen3-VL-4B int8 + `v3-mlp` | 4.83 GB | 503 MB | **5.3 GB** |
|
| 31 |
-
| Qwen3-VL-4B int8 + `v3` | 4.83 GB | 52 MB | **4.9 GB** |
|
| 32 |
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
| | |
|
| 36 |
-
|---|---|
|
| 37 |
-
| diffusion model | `minimax_h3_fl2va_pruned_int8_convrot` |
|
| 38 |
-
| turbo LoRA | `minimax_h3_fl2v_turbo_8step_v1.0_comfyui_bf16`, strength 1.0 |
|
| 39 |
-
| sampler / scheduler | `res_multistep` / `simple` |
|
| 40 |
-
| steps | 8 |
|
| 41 |
-
| seed | 42 |
|
| 42 |
-
| resolution | 16:9 at 0.8 MP, upscaled 2× by RTX Video Super Resolution ULTRA |
|
| 43 |
-
| frames | 192 at 24 fps — 8.00 s |
|
| 44 |
-
| video VAE | `minimax_h3_video_vae_int8_convrot` |
|
| 45 |
-
| audio VAE | `minimax_h3_audio_vae_fp32` |
|
| 46 |
-
|
| 47 |
-
**Only the projection changes between the five.** Everything else is identical, and on one machine the pipeline is deterministic — running the same configuration twice gives byte-identical decoded video and audio, verified by MD5 — so every difference you see comes from the projection and nothing else.
|
| 48 |
-
|
| 49 |
-
### You will not reproduce these files, and that is expected
|
| 50 |
-
|
| 51 |
-
Run the demo prompt with seed 42 on your own machine and you will get the same scene, not the same file. **The result depends on the model of GPU the encoder runs on.**
|
| 52 |
-
|
| 53 |
-
This came out of an unrelated test — checking that three loading modes gave the same output — and the cards happened to be at hand. Four of them is not a study, and none of this was the point of the exercise; it is written down because it would otherwise look like something is broken. Same prompt, same seed, same everything else:
|
| 54 |
-
|
| 55 |
-
| card | decoded video MD5 |
|
| 56 |
-
|---|---|
|
| 57 |
-
| RTX 4070 | `1daf9be3…` |
|
| 58 |
-
| RTX 3090 | `0a415964…` |
|
| 59 |
-
| RTX 3060 | `dcb2f965…` |
|
| 60 |
-
| RTX 4070 Ti SUPER | `b3b185e7…` |
|
| 61 |
-
|
| 62 |
-
Four cards, four results. **Two different RTX 3090s gave byte-identical output**, so it is the model that decides, not the individual card — and not the architecture either, since the 3060 and the 3090 are both Ampere and disagree.
|
| 63 |
-
|
| 64 |
-
The cause is small and the consequence is not. Encoding the same prompt on two cards gives conditioning that agrees to a **relative error of 7 × 10⁻⁷** — cosine 1.00000000, largest single-component difference 0.002. Different numbers of compute units mean different reduction orders, so floating-point additions do not happen in the same sequence. Eight denoising steps turn that into a different piece of furniture, or a wristwatch that is there on one card and absent on another. That watch is nowhere in the prompt, which is exactly why it is free to move.
|
| 65 |
-
|
| 66 |
-
So: on one machine, with one card, everything here is reproducible to the bit — that is what makes the five-way comparison above meaningful. Across machines, expect the same scene with different details. This is a property of the diffusion model and its sampler, not of the projections: the reference 32B behaves identically.
|
| 67 |
-
|
| 68 |
-
The full prompt is in [`demo/chess-prompt.txt`](demo/chess-prompt.txt), the settings above in machine-readable form in [`demo/generation-settings.json`](demo/generation-settings.json), and the five renders are in [`demo/`](demo) one by one if you want to step through them.
|
| 69 |
-
|
| 70 |
-
**What to look at.** Everything the prompt states is there, on all five: the seated pose, the red dress, the white pieces on her side, the two captured black pawns, the cat, the straw hat, the laundry — and her knee, asked for three times and ending on a sentence of its own, *"Her knee never stops bouncing."* A continuous involuntary motion with no narrative purpose is the clearest single sign that a projection carried what was written, and it carries on the plain matrices too.
|
| 71 |
-
|
| 72 |
-
**One thing none of the five gets right, the 32B included:** she lifts a knight and does not set it back on the same square, and on the 8B residual there is no knight on the board at all. Object permanence behind an occluding hand, on a grid of sixty-four identical squares, is a limit of the video model rather than of the conditioning.
|
| 73 |
-
|
| 74 |
-
**And the terrace is furnished differently from one render to the next — that is not infidelity.** The prompt asks for a densely lived-in terrace without anchoring most of it: the cat is *"stretched out asleep in the sun"*, and nothing says where. What is left open, the model invents, and it invents differently depending on **the projection, the seed, and the model of GPU** — all three act on that same free space and none of them touches what was written. Two renders side by side give the impression of a different seed; that impression is what an unconstrained description looks like.
|
| 75 |
-
|
| 76 |
-
> ⚠️ **Proof of concept — working, but a proof of concept.** It runs and produces good video, and every number below was measured on real hardware. Built and tested on a single setup (Windows 11, NVIDIA, ComfyUI 0.31.0) with deliberately limited exploration.
|
| 77 |
|
| 78 |
-
|
| 79 |
|
| 80 |
-
|
| 81 |
-
|
| 82 |
|
| 83 |
-
|
|
|
|
| 84 |
|
| 85 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 86 |
|
| 87 |
-
|
|
|
|
| 88 |
|
| 89 |
-
|
|
|
|
| 90 |
|
| 91 |
-
|
| 92 |
|
| 93 |
-
##
|
| 94 |
|
| 95 |
-
|
| 96 |
|
| 97 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 98 |
|
| 99 |
-
|
| 100 |
-
|
| 101 |
-
|
| 102 |
-
| 8B residual, after | 0.8578 | 0.9605 |
|
| 103 |
-
| 8B matrix, before | 0.7845 | 0.8926 |
|
| 104 |
-
| 8B matrix, after | 0.8457 | 0.9361 |
|
| 105 |
|
| 106 |
-
|
|
|
|
| 107 |
|
| 108 |
-
**
|
|
|
|
|
|
|
|
|
|
| 109 |
|
| 110 |
-
|
|
|
|
| 111 |
|
| 112 |
-
|
| 113 |
|
| 114 |
-
|
| 115 |
-
|---|---|
|
| 116 |
-
| `mmh3-8b-ClipProj-v3-mlp` | 0.9449 |
|
| 117 |
-
| v2 `mmh3-8b-ClipProj-celeb-mlp` | 0.9393 |
|
| 118 |
-
| `mmh3-4b-ClipProj-v3-mlp` | 0.9381 |
|
| 119 |
-
| v2 `mmh3-4b-ClipProj-celeb-mlp` | 0.9293 |
|
| 120 |
-
| `mmh3-8b-ClipProj-v3` | 0.9289 |
|
| 121 |
-
| `mmh3-4b-ClipProj-v3` | 0.9193 |
|
| 122 |
|
| 123 |
-
|
|
|
|
|
|
|
| 124 |
|
| 125 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 126 |
|
| 127 |
-
|
| 128 |
|
| 129 |
-
##
|
| 130 |
|
| 131 |
-
|
|
|
|
|
|
|
| 132 |
|
| 133 |
```
|
| 134 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 135 |
```
|
| 136 |
|
| 137 |
-
|
| 138 |
|
| 139 |
-
|
|
|
|
| 140 |
|
| 141 |
-
|
| 142 |
-
|
| 143 |
-
|
| 144 |
-
|
| 145 |
-
**Start with `mmh3-8b-ClipProj-v3-mlp` if you have the VRAM, `mmh3-4b-ClipProj-v3-mlp` otherwise.** Both need node 0.1.13 or later.
|
| 146 |
-
|
| 147 |
-
| File | Encoder | Structure | Size | vs 32B |
|
| 148 |
-
|---|---|---|---|---|
|
| 149 |
-
| `mmh3-4b-ClipProj-v3-mlp` | any Qwen3-VL-4B | non-linear | 503 MB | 0.9381 |
|
| 150 |
-
| `mmh3-8b-ClipProj-v3-mlp` | any Qwen3-VL-8B | non-linear | 604 MB | **0.9449** |
|
| 151 |
-
| `mmh3-4b-ClipProj-v3` | any Qwen3-VL-4B | matrix only | 26 MB | 0.9193 |
|
| 152 |
-
| `mmh3-8b-ClipProj-v3` | any Qwen3-VL-8B | matrix only | 42 MB | 0.9289 |
|
| 153 |
-
| `mmh3-ClipProj-control-zero` | — | control, run it once | 52 MB | — |
|
| 154 |
-
| `mmh3-ClipProj-control-identity` | — | control, run it once | 52 MB | — |
|
| 155 |
-
|
| 156 |
-
The v2 files — `mmh3-4b-ClipProj-celeb-mlp` and the seven beside it — are kept and still work. They were calibrated against a modified 32B and against a corpus containing no image tokens, so prefer v3.
|
| 157 |
-
|
| 158 |
-
The last column is the one number measured identically for every row: same stock 32B, same prompt, cosine averaged token by token. It is also written inside each file as `cos_prompt_reference`.
|
| 159 |
-
|
| 160 |
-
The `-v3-mlp` files are larger than the v2 residuals — 503 and 604 MB against 304 and 386 — because the hidden width went from 16 384 to 32 768. That is the whole reason for the extra download.
|
| 161 |
-
|
| 162 |
-
Every matrix works on **any variant of its own size**: the measured cosine gap between a bf16-calibrated matrix applied to an abliterated fp8 encoder is 0.0023. You do not need the exact checkpoint a matrix was calibrated on. The 8B matrices need an 8B encoder though — 4096 input dimensions instead of 2560 — and the node checks the width and refuses a mismatch.
|
| 163 |
-
|
| 164 |
-
## Named people
|
| 165 |
-
|
| 166 |
-
**This is what changed in 0.1.3, and it was a corpus problem.**
|
| 167 |
-
|
| 168 |
-
The calibration corpus named a person on about 70 lines out of 8632, roughly 0.02 % of the training tokens. The directions of the hidden space that carry an identity were therefore constrained by nothing at all, and the fit put whatever minimised the error on landscape descriptions there. Named people came out as somebody else.
|
| 169 |
-
|
| 170 |
-
The `-celeb` matrices add 500 people, ranked by popularity, with five short prompts and two long ones each. What it buys and what it costs:
|
| 171 |
-
|
| 172 |
-
| | name tokens | rest of the sentence | general test set |
|
| 173 |
-
|---|---|---|---|
|
| 174 |
-
| without | 0.8265 | 0.9358 | 0.7944 |
|
| 175 |
-
| with | **0.8844** | **0.9516** | 0.7930 |
|
| 176 |
-
|
| 177 |
-
Seven thousandths of cosine on the general corpus, for six points on the tokens that carry an identity. The rest of the sentence improves too, because the celebrity prompts are short and the general corpus had nothing under fifteen words.
|
| 178 |
-
|
| 179 |
-
Two findings that decide how far this is worth pushing.
|
| 180 |
-
|
| 181 |
-
**Two contexts per person are enough.** Measured on contexts held out for people the matrix had seen: 0.9875 at two, 0.9945 at five, 0.9986 at twenty. Forty is a waste.
|
| 182 |
-
|
| 183 |
-
**Five hundred names generalise to names never seen.** A held-out band at popularity ranks 501 to 540, absent from every calibration, reconstructs at 0.8795 against 0.8844 for the covered ones. Covering 500 people does not teach 500 names; it teaches the map how to handle that region of the space. Going to several thousand would buy very little.
|
| 184 |
|
| 185 |
-
|
|
|
|
| 186 |
|
| 187 |
-
##
|
| 188 |
|
| 189 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 190 |
|
| 191 |
-
|
| 192 |
|
| 193 |
-
|
|
|
|
| 194 |
|
| 195 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 196 |
|
| 197 |
-
**
|
|
|
|
|
|
|
| 198 |
|
| 199 |
-
|
|
|
|
|
|
|
|
|
|
| 200 |
|
| 201 |
-
##
|
| 202 |
|
| 203 |
-
|
|
|
|
|
|
|
|
|
|
| 204 |
|
| 205 |
-
|
| 206 |
|
| 207 |
-
|
| 208 |
-
|
| 209 |
-
|
| 210 |
-
| matrix + residual | 0.7944 | 0.7970 |
|
| 211 |
-
| matrix, names covered | 0.7095 | 0.7466 |
|
| 212 |
-
| matrix + residual, names covered | 0.7930 | **0.8037** |
|
| 213 |
|
| 214 |
-
|
| 215 |
|
| 216 |
-
|
| 217 |
|
| 218 |
-
|
|
|
|
|
|
|
| 219 |
|
| 220 |
-
|
|
|
|
|
|
|
| 221 |
|
| 222 |
-
|
|
|
|
| 223 |
|
| 224 |
-
|
|
|
|
|
|
|
|
|
|
| 225 |
|
| 226 |
## Run the controls first
|
| 227 |
|
| 228 |
-
|
|
|
|
| 229 |
|
| 230 |
| Matrix | Output for *"a red ball on a wood table"* |
|
| 231 |
|---|---|
|
|
@@ -233,60 +221,53 @@ The two control matrices exist to prove the learned matrix is doing the work rat
|
|
| 233 |
| `mmh3-ClipProj-control-identity` | a golden object in flames — unusable |
|
| 234 |
| a learned matrix | the red ball on a wood table |
|
| 235 |
|
| 236 |
-
`‖W_identity‖ = 50.6` against `‖W_learned‖ = 52.4` — near-identical energy, so the difference is
|
| 237 |
-
|
| 238 |
-
|
| 239 |
-
|
| 240 |
-
## What is in obsolete/
|
| 241 |
-
|
| 242 |
-
The previous matrices, kept because a comparison posted on r/StableDiffusion ran on them and the links have to keep working. They have no name coverage and are calibrated on a corpus thirty times smaller. There is no reason to prefer them.
|
| 243 |
|
| 244 |
-
|
| 245 |
|
| 246 |
-
|
|
|
|
| 247 |
|
| 248 |
-
|
|
|
|
| 249 |
|
| 250 |
-
|
| 251 |
-
|
| 252 |
-
|
| 253 |
-
```
|
| 254 |
-
|
| 255 |
-
They are the same function. Unregularised least squares is invariant to an invertible linear transform of the targets, so fitting in one space and mapping back recovers the same map; only the ridge penalty breaks that invariance, and with 37 851 training tokens against λ = 1000 it barely binds. The entire gain was an artefact of measuring in a different space.
|
| 256 |
-
|
| 257 |
-
*The idea came from u/stddealer on r/StableDiffusion, and it was a good one. The measurement is on me: I published the cosine before checking whether the matrix had changed at all.*
|
| 258 |
|
| 259 |
-
|
|
|
|
| 260 |
|
| 261 |
-
**Quantisation costs facts
|
| 262 |
|
| 263 |
-
**Masks defeat identity
|
| 264 |
-
|
| 265 |
-
**Counting is unreliable, and not because of the projection.** Ask for three of something and you get four, on the 32B too. Enumerating works better than announcing a number.
|
| 266 |
|
| 267 |
## Required models
|
| 268 |
|
| 269 |
| Role | Model |
|
| 270 |
|---|---|
|
| 271 |
| Diffusion model + VAEs | [Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3) |
|
| 272 |
-
| Text encoder, 4B |
|
| 273 |
-
| Text encoder, 8B | any ComfyUI-format Qwen3-VL-8B
|
| 274 |
|
| 275 |
The 32B text encoder is **no longer needed** — that is the entire point.
|
| 276 |
|
| 277 |
## Licence and responsibility
|
| 278 |
|
| 279 |
-
These matrices are
|
| 280 |
-
|
| 281 |
-
They are derived from the activations of both models, and their legal status is unclear. They are provided as-is, for research, with no claim of ownership over anything derived from the underlying models.
|
| 282 |
-
|
| 283 |
-
- **Qwen3-VL** is published by Alibaba under **Apache 2.0**. Read and comply with its terms and acceptable-use policy.
|
| 284 |
-
- **MiniMax H3** ships under a **custom licence**. Read it before any use, particularly commercial.
|
| 285 |
|
| 286 |
-
|
|
|
|
| 287 |
|
| 288 |
-
|
|
|
|
| 289 |
|
| 290 |
## Credits
|
| 291 |
|
| 292 |
-
Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number quoted was measured on
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
- text-to-video
|
| 7 |
- qwen3-vl
|
| 8 |
- text-encoder
|
| 9 |
+
- multilingual
|
| 10 |
base_model:
|
| 11 |
- Comfy-Org/MiniMax-H3
|
| 12 |
- Qwen/Qwen3-VL-4B-Instruct
|
| 13 |
+
- Qwen/Qwen3-VL-8B-Instruct
|
| 14 |
library_name: comfyui
|
| 15 |
---
|
| 16 |
|
| 17 |
# ClipProj — MiniMax H3 conditioning from a Qwen3-VL-4B or 8B
|
| 18 |
|
| 19 |
**Projection matrices that let a small Qwen3-VL replace the Qwen3-VL-32B text encoder of MiniMax H3.**
|
| 20 |
+
**15.0 GB → 4.6 GB**, with no change to the diffusion model, the VAEs or the sampler.
|
| 21 |
|
| 22 |
+
<video controls width="360" src="https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4"></video>
|
| 23 |
|
| 24 |
+
*Eleven languages, 88 seconds. For each one, the **smallest** file that matches the 32B — not the best
|
| 25 |
+
one. Nine of the eleven run on a 4B.
|
| 26 |
+
[Direct link](https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4)*
|
| 27 |
|
| 28 |
+
> ⚠️ **The video looks rough, and that is on purpose.** It is rendered at 0.3 MP with 6 sampling steps —
|
| 29 |
+
> the settings that made 297 renders affordable — then upscaled. The audio is tinny for the same reason,
|
| 30 |
+
> **identically so on the 32B**, with the same 19 dB dip between 1 and 3 kHz. This is a pronunciation test,
|
| 31 |
+
> not a showcase: it exists to let you hear *which words come out*, not how pretty the result is. Render at
|
| 32 |
+
> your own settings and it will look like MiniMax H3 normally looks.
|
| 33 |
|
| 34 |
+
Requires the custom node **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**,
|
| 35 |
+
version 0.1.13 or later. The v3.1 files need no code change — same base as v3.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 36 |
|
| 37 |
+
---
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 38 |
|
| 39 |
+
## Read this before the tables: these are metrics, not verdicts
|
| 40 |
|
| 41 |
+
**I do not speak these eleven languages.** I cannot tell you whether a render sounds right, and I have not
|
| 42 |
+
asked anyone who can. No native speaker has listened to any of the 297 speech renders behind this page.
|
| 43 |
|
| 44 |
+
Every figure below is a **distance between two automatic transcriptions** — what one machine wrote down
|
| 45 |
+
from the reference, against what it wrote down from the projection. That is all it is.
|
| 46 |
|
| 47 |
+
**What that captures.** Whether the same words and the same sounds come out. Two instruments are used
|
| 48 |
+
because each is wrong in a known direction: **Whisper** has a language model inside and corrects a slurred
|
| 49 |
+
word into the most probable real one, so it *under*-reports defects — a lower bound. **ZIPA** has no
|
| 50 |
+
lexical decoder and counts every shift in realisation as an error, so it *over*-reports — an upper bound.
|
| 51 |
+
What a listener would notice lies between them.
|
| 52 |
|
| 53 |
+
**What it does not capture.** Prosody, rhythm, timbre, naturalness. A file scoring 98.8 here could still
|
| 54 |
+
sound foreign to someone who speaks the language.
|
| 55 |
|
| 56 |
+
**So read a score as "close to the 32B, according to this instrument" — never as "good".** If you speak one
|
| 57 |
+
of these languages, your ear outranks every table here, and I would genuinely like to hear what it says.
|
| 58 |
|
| 59 |
+
---
|
| 60 |
|
| 61 |
+
## Files — take a v3.1
|
| 62 |
|
| 63 |
+
Put them in `ComfyUI/models/clip_projections/`.
|
| 64 |
|
| 65 |
+
| File | Encoder | Head | Projection | Encoder + projection |
|
| 66 |
+
|---|---|---|---|---|
|
| 67 |
+
| **`mmh3-4b-ClipProj-v3.1`** | any Qwen3-VL-4B | ridge | 26 MB | **4.6 GB** |
|
| 68 |
+
| `mmh3-4b-ClipProj-v3.1-mlp` | any Qwen3-VL-4B | residual | 481 MB | 5.1 GB |
|
| 69 |
+
| `mmh3-8b-ClipProj-v3.1` | any Qwen3-VL-8B | ridge | 41 MB | 9.6 GB |
|
| 70 |
+
| `mmh3-8b-ClipProj-v3.1-mlp` | any Qwen3-VL-8B | residual | 577 MB | 10.1 GB |
|
| 71 |
|
| 72 |
+
**Start with `mmh3-4b-ClipProj-v3.1`.** Nine of the eleven benchmarked languages run on it, and nothing
|
| 73 |
+
distinguishes it from the 32B in image generation. The 8B earns its extra 5 GB on Arabic, French, and
|
| 74 |
+
proper nouns generally.
|
|
|
|
|
|
|
|
|
|
| 75 |
|
| 76 |
+
The 8B matrices expect 4096 input dimensions instead of 2560; the node checks the width and refuses a
|
| 77 |
+
mismatch. Every matrix works on any variant of its own size.
|
| 78 |
|
| 79 |
+
**One format detail.** The two `-mlp` files carry **no `W` tensor** — they were trained without a linear
|
| 80 |
+
path, so the residual network carries everything, and a matrix of zeros would cost 26 MB plus a 2560×5120
|
| 81 |
+
matmul per token to add nothing. The node treats `W` as optional and reports `| residual only`. This is
|
| 82 |
+
expected, not a truncated download.
|
| 83 |
|
| 84 |
+
The earlier `-celeb` and v3 files remain available. There is no reason to prefer them: the benchmark
|
| 85 |
+
separates v3 from v3.1 cleanly, and in the same direction on every metric.
|
| 86 |
|
| 87 |
+
---
|
| 88 |
|
| 89 |
+
## What changed in v3.1: giving every script its share
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 90 |
|
| 91 |
+
v3 was calibrated on a corpus that was overwhelmingly English, with the other languages bolted on
|
| 92 |
+
afterwards as a top-up. v3.1 **adds text and tagged prompts until every writing system carries roughly
|
| 93 |
+
comparable weight** — English excepted, since the prompt format itself is English.
|
| 94 |
|
| 95 |
+
| Script | Languages | Tagged prompts | Raw text | Share of corpus |
|
| 96 |
+
|---|---|---|---|---|
|
| 97 |
+
| Latin — base | English: original corpus, image and register lots | *base* | — | **68.3 %** |
|
| 98 |
+
| Han | zh | ✓ | ✓ | 6.7 % |
|
| 99 |
+
| Hangul | ko | ✓ | ✓ | 6.6 % |
|
| 100 |
+
| Latin, accented | fr | ✓ | ✓ | 6.3 % |
|
| 101 |
+
| Arabic | ar | ✓ | ✓ **(new)** | 4.1 % |
|
| 102 |
+
| Latin | es, de, it, pt | ✓ | — | 1.3 % each |
|
| 103 |
+
| Cyrillic | ru | ✓ | — | 1.3 % |
|
| 104 |
+
|
| 105 |
+
The rule behind those numbers: **a script that inherits nothing from Latin needs raw text**; a Latin script
|
| 106 |
+
only needs tagged prompts, because the alphabet is already covered. The one genuinely new lot is raw
|
| 107 |
+
Arabic — 550 000 characters, the last non-Latin script still living on tagged prompts alone.
|
| 108 |
+
|
| 109 |
+
Training also restarted from scratch rather than topping up: a network keeps the order it learned in, and
|
| 110 |
+
lowering the learning rate on a top-up only arbitrates between preserving and correcting. Architecture and
|
| 111 |
+
hyper-parameters are identical to v3 — `hidden 32768`, `depth 1`, `tap 24`, `lr 1e-3` — so what the
|
| 112 |
+
benchmark compares is the corpus, not the recipe.
|
| 113 |
|
| 114 |
+
---
|
| 115 |
|
| 116 |
+
## The benchmark
|
| 117 |
|
| 118 |
+
Everything below comes from **three seeds** — 42, 100 000 and 100 000 000 — because a single draw cannot
|
| 119 |
+
separate a real gap from chance. All of it is in [`bench3.1/`](./bench3.1): 297 speech renders, 405 image
|
| 120 |
+
renders, the raw CSVs and the montage.
|
| 121 |
|
| 122 |
```
|
| 123 |
+
bench3.1/
|
| 124 |
+
README.md this measurement report in full
|
| 125 |
+
audio/<lang>/ 297 FLAC — 9 conditionings × 11 languages × 3 seeds
|
| 126 |
+
video/langues/<lang>/ 99 renders, one per conditioning
|
| 127 |
+
video/ the eleven-language montage
|
| 128 |
+
images/brutes/ 75 PNG — 15 scenes × 5 conditionings
|
| 129 |
+
images/planches/ 15 comparison sheets, five renders side by side
|
| 130 |
+
mesures/ phonemes, words, image cosines, scores, prompts
|
| 131 |
```
|
| 132 |
|
| 133 |
+
### The reference is not perfection
|
| 134 |
|
| 135 |
+
A cosine of 0.79 or "23 character errors" has no scale, and zero errors is not the target either, because
|
| 136 |
+
**the 32B does not reproduce itself**. Change nothing but the seed:
|
| 137 |
|
| 138 |
+
| | 32B against itself |
|
| 139 |
+
|---|---|
|
| 140 |
+
| Speech | **5.8 phonemes out of 75** (7.8 %) |
|
| 141 |
+
| Image | **0.9552** SigLIP2 cosine (floor 0.5313) |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 142 |
|
| 143 |
+
That gap is the unit. Everything is normalised so **32B = 100**: at 100, swapping the encoder moves the
|
| 144 |
+
output as much as changing the seed does.
|
| 145 |
|
| 146 |
+
### Results
|
| 147 |
|
| 148 |
+
| Conditioning | Speech | ± | Image | Prompt | Marseille |
|
| 149 |
+
|---|---|---|---|---|---|
|
| 150 |
+
| **32B** *(reference)* | **100.0** | — | **100.0** | **100.0** | 3/3 |
|
| 151 |
+
| `8b-ClipProj-v3.1` | **98.8** | ±1.4 | 100.0 | 100.4 | 3/3 |
|
| 152 |
+
| `8b-ClipProj-v3.1-mlp` | 98.2 | ±1.2 | 101.4 | **102.3** | 3/3 |
|
| 153 |
+
| `4b-ClipProj-v3.1` | 97.9 | ±1.0 | 100.1 | 99.4 | 0/3 |
|
| 154 |
+
| `4b-ClipProj-v3.1-mlp` | 97.8 | ±1.1 | 100.9 | 100.3 | 1/3 |
|
| 155 |
+
| *the four v3 files* | *93.2 – 95.6* | | *99.4 – 102.4* | *99.0 – 100.7* | *0–3/3* |
|
| 156 |
|
| 157 |
+
`±` is the spread across the three seeds. **Two files separated by less than that are not separated.**
|
| 158 |
|
| 159 |
+
The raw counts, three units never added together — PER over three seeds, WER and CER as Whisper hears
|
| 160 |
+
them, single-seed:
|
| 161 |
|
| 162 |
+
| | PER | | WER | CER |
|
| 163 |
+
|---|---|---|---|---|
|
| 164 |
+
| *32B against itself* | *≈193 / 2469* | *7.8 %* | — | — |
|
| 165 |
+
| `8b-v3.1` | 211 / 2469 | 8.5 % | 10 / 174 | 14 / 869 |
|
| 166 |
+
| `8b-v3.1-mlp` | 222 / 2469 | 9.0 % | 15 / 174 | 28 / 869 |
|
| 167 |
+
| `4b-v3.1-mlp` | 230 / 2469 | 9.3 % | 13 / 174 | 23 / 869 |
|
| 168 |
+
| `4b-v3.1` | 232 / 2469 | 9.4 % | 15 / 174 | 30 / 869 |
|
| 169 |
+
| the four v3 | 281–341 / 2469 | 11.4–13.8 % | 17–23 / 174 | 29–46 / 869 |
|
| 170 |
|
| 171 |
+
**What separates: the corpus.** The four v3.1 files land within one point of each other while ranging from
|
| 172 |
+
4.6 to 10.1 GB. The four v3 sit a clear notch below at identical sizes. A 4B v3.1 beats an 8B v3 by four
|
| 173 |
+
points while weighing half as much.
|
| 174 |
|
| 175 |
+
**What does not separate: everything else.** Not 4B against 8B on general pronunciation, not ridge against
|
| 176 |
+
residual, and nothing at all in image generation — all nine conditionings, v3 included, sit between 99.4
|
| 177 |
+
and 102.4 there, where the standard deviation within a single model is three to five times the entire
|
| 178 |
+
spread between models.
|
| 179 |
|
| 180 |
+
### The one place 4B and 8B part company
|
| 181 |
|
| 182 |
+
The French line says *"la lumière de **Marseille**"*. Six phonemes out of seventy — the error rate drowns
|
| 183 |
+
them, the ear does not. Across three seeds, the 32B and three 8B files say it every time; the best 4B says
|
| 184 |
+
it once out of three. Treat that for what it is: **one proper noun, in one language out of eleven**. It is
|
| 185 |
+
not a ranking criterion, but if your prompts lean on names, test both.
|
| 186 |
|
| 187 |
+
### Full report
|
| 188 |
|
| 189 |
+
[`bench3.1/README.md`](./bench3.1/README.md) carries the whole thing: per-language tables, the raw-text
|
| 190 |
+
correlation that does *not* hold, why the phoneme metric misleads in Portuguese and Russian, the
|
| 191 |
+
self-consistency measurements, and every limitation I know of.
|
|
|
|
|
|
|
|
|
|
| 192 |
|
| 193 |
+
---
|
| 194 |
|
| 195 |
+
## What this is, technically
|
| 196 |
|
| 197 |
+
MiniMax H3 conditions on a Qwen3-VL-32B truncated to 50 layers — 15.0 GB in NVFP4 — solely to turn a
|
| 198 |
+
prompt into a `[seq, 5120]` tensor. This repository provides a learned map so a much smaller Qwen3-VL can
|
| 199 |
+
produce the same conditioning:
|
| 200 |
|
| 201 |
+
```
|
| 202 |
+
cond = ((h - mean_in) / std_in) @ W * std_out + mean_out
|
| 203 |
+
```
|
| 204 |
|
| 205 |
+
plus, in the `-mlp` files, the output of a residual network fed the same standardised input — and in the
|
| 206 |
+
v3.1 residuals, *only* that network.
|
| 207 |
|
| 208 |
+
It works because every Qwen3-VL shares the **same tokenizer** (151936 tokens): a prompt yields the same
|
| 209 |
+
tokens at the same positions in both models, so a position-by-position mapping between hidden states can
|
| 210 |
+
be learned. The matrix is fitted by plain **ridge regression** — no gradients, no epochs. Only the residual
|
| 211 |
+
network is trained.
|
| 212 |
|
| 213 |
## Run the controls first
|
| 214 |
|
| 215 |
+
Two control matrices prove the learned matrix is doing the work rather than the diffusion model. Same
|
| 216 |
+
prompt, same seed, only the matrix changes:
|
| 217 |
|
| 218 |
| Matrix | Output for *"a red ball on a wood table"* |
|
| 219 |
|---|---|
|
|
|
|
| 221 |
| `mmh3-ClipProj-control-identity` | a golden object in flames — unusable |
|
| 222 |
| a learned matrix | the red ball on a wood table |
|
| 223 |
|
| 224 |
+
`‖W_identity‖ = 50.6` against `‖W_learned‖ = 52.4` — near-identical energy, so the difference is
|
| 225 |
+
structural, not a matter of scale. **If the identity control ever looks fine, the learned matrix adds
|
| 226 |
+
nothing, and you want to know that before trusting it.**
|
|
|
|
|
|
|
|
|
|
|
|
|
| 227 |
|
| 228 |
+
## Limitations
|
| 229 |
|
| 230 |
+
**Three seeds fix the order of magnitude of the noise, not its tail.** Any gap under one point of score is
|
| 231 |
+
not a result.
|
| 232 |
|
| 233 |
+
**The cosine is blind to countable attributes.** A whole loaf and a halved loaf, same crust, same paper,
|
| 234 |
+
same light, give the same vector to the fourth decimal. Image equivalence here means *global appearance*.
|
| 235 |
|
| 236 |
+
**The phoneme metric is unreliable in Portuguese, Russian and Korean** — not the speech itself. The
|
| 237 |
+
reference drifts by 17.0, 10.7 and 11.3 phonemes there between seeds, while Whisper transcribes the same
|
| 238 |
+
renders with zero to three character errors.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 239 |
|
| 240 |
+
**Speech quality is deliberately poor**, as the video shows — 6 to 8 steps, identically for the 32B. The
|
| 241 |
+
benchmark measures correctness of pronunciation, not fidelity of reproduction.
|
| 242 |
|
| 243 |
+
**Quantisation costs facts**, which is the most likely explanation for the proper-noun gap.
|
| 244 |
|
| 245 |
+
**Masks defeat identity**, and **counting is unreliable** — both true on the 32B too.
|
|
|
|
|
|
|
| 246 |
|
| 247 |
## Required models
|
| 248 |
|
| 249 |
| Role | Model |
|
| 250 |
|---|---|
|
| 251 |
| Diffusion model + VAEs | [Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3) |
|
| 252 |
+
| Text encoder, 4B | any ComfyUI-format Qwen3-VL-4B |
|
| 253 |
+
| Text encoder, 8B | any ComfyUI-format Qwen3-VL-8B |
|
| 254 |
|
| 255 |
The 32B text encoder is **no longer needed** — that is the entire point.
|
| 256 |
|
| 257 |
## Licence and responsibility
|
| 258 |
|
| 259 |
+
MIT, like the node. These matrices are derived from the activations of both models and their legal status
|
| 260 |
+
is unclear; they are provided as-is, for research.
|
|
|
|
|
|
|
|
|
|
|
|
|
| 261 |
|
| 262 |
+
- **Qwen3-VL** — Alibaba, Apache 2.0. Read its terms and acceptable-use policy.
|
| 263 |
+
- **MiniMax H3** — custom licence. Read it before any use, particularly commercial.
|
| 264 |
|
| 265 |
+
Not affiliated with, endorsed by, or connected to Alibaba / Qwen, MiniMax, or Comfy Org. You remain
|
| 266 |
+
responsible for what you generate and for complying with the licences of every model you load.
|
| 267 |
|
| 268 |
## Credits
|
| 269 |
|
| 270 |
+
Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number quoted was measured on this hardware,
|
| 271 |
+
never estimated. Where a prediction lost to a measurement, the measurement won and the text was rewritten
|
| 272 |
+
— which happened several times in this release, the largest being a single-seed ranking of the v3.1 files
|
| 273 |
+
that dissolved entirely once the 32B's own variance was known.
|
bench3.1/README.md
ADDED
|
@@ -0,0 +1,494 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: mit
|
| 3 |
+
tags:
|
| 4 |
+
- comfyui
|
| 5 |
+
- minimax-h3
|
| 6 |
+
- text-to-video
|
| 7 |
+
- qwen3-vl
|
| 8 |
+
- text-encoder
|
| 9 |
+
- multilingual
|
| 10 |
+
base_model:
|
| 11 |
+
- Comfy-Org/MiniMax-H3
|
| 12 |
+
- Qwen/Qwen3-VL-4B-Instruct
|
| 13 |
+
- Qwen/Qwen3-VL-8B-Instruct
|
| 14 |
+
library_name: comfyui
|
| 15 |
+
---
|
| 16 |
+
|
| 17 |
+
# ClipProj v3.1 — measured against the 32B's own variance
|
| 18 |
+
|
| 19 |
+
**Four projection matrices that let a Qwen3-VL-4B or 8B replace the Qwen3-VL-32B text encoder of MiniMax H3.**
|
| 20 |
+
|
| 21 |
+
**15.0 GB → 4.6 GB**, with no change to the diffusion model, the VAEs or the sampler.
|
| 22 |
+
|
| 23 |
+
This release is not about a new architecture. It is about finally knowing **how good these things are**, because the previous numbers could not tell me. This card is mostly the measurement, and the measurement changed three of my own conclusions.
|
| 24 |
+
|
| 25 |
+
Requires the custom node: **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
|
| 26 |
+
|
| 27 |
+
---
|
| 28 |
+
|
| 29 |
+
## Read this before the tables: these are metrics, not verdicts
|
| 30 |
+
|
| 31 |
+
**I do not speak these eleven languages.** I cannot tell you whether a render sounds right, and I have not
|
| 32 |
+
asked anyone who can. No native speaker has listened to any of the 297 speech renders on this page.
|
| 33 |
+
|
| 34 |
+
So nothing below is a judgement of quality. Every figure is a **distance between two automatic
|
| 35 |
+
transcriptions** — what one machine wrote down from the reference, against what it wrote down from the
|
| 36 |
+
projection. That is all it is, and it is worth being explicit about what that does and does not capture:
|
| 37 |
+
|
| 38 |
+
**What the numbers do capture.** Whether the same words and the same sounds come out. Two instruments are
|
| 39 |
+
used precisely because each is wrong in a known direction: **Whisper** has a language model inside and
|
| 40 |
+
corrects a slurred word into the most probable real one, so it *under*-reports pronunciation defects — a
|
| 41 |
+
lower bound. **ZIPA** has no lexical decoder at all and counts every shift in realisation as an error, so
|
| 42 |
+
it *over*-reports — an upper bound. What a listener would notice lies between them, and neither number
|
| 43 |
+
alone is the answer.
|
| 44 |
+
|
| 45 |
+
**What they do not capture.** Prosody, rhythm, timbre, naturalness — everything that makes speech sound
|
| 46 |
+
native rather than merely correct. A render scoring 98.8 here could still sound foreign to someone who
|
| 47 |
+
speaks the language. These metrics cannot see that, and neither can I.
|
| 48 |
+
|
| 49 |
+
**So read a score as "close to the 32B, according to this instrument"** — never as "good". If you speak
|
| 50 |
+
one of these languages, your ear outranks every table below, and I would genuinely like to hear what it
|
| 51 |
+
tells you.
|
| 52 |
+
|
| 53 |
+
---
|
| 54 |
+
|
| 55 |
+
## What changed in v3.1: giving every script its share
|
| 56 |
+
|
| 57 |
+
v3 was calibrated on a corpus that was overwhelmingly English, with the other languages bolted on
|
| 58 |
+
afterwards as a top-up. v3.1 **adds text and tagged prompts until every writing system carries roughly
|
| 59 |
+
comparable weight** — English excepted, because the prompt format itself is English: the sections, the
|
| 60 |
+
tags and the descriptions are all written in it, so it stays the majority no matter what.
|
| 61 |
+
|
| 62 |
+
Measured share of the v3.1 corpus:
|
| 63 |
+
|
| 64 |
+
| Script | Languages | Tagged prompts | Raw text | Share |
|
| 65 |
+
|---|---|---|---|---|
|
| 66 |
+
| Latin — base | English: the original corpus, image lots and register lots | *base* | — | **68.3 %** |
|
| 67 |
+
| Han | zh | ✓ | ✓ | 6.7 % |
|
| 68 |
+
| Hangul | ko | ✓ | ✓ | 6.6 % |
|
| 69 |
+
| Latin, accented | fr | ✓ | ✓ | 6.3 % |
|
| 70 |
+
| Arabic | ar | ✓ | ✓ **(new)** | 4.1 % |
|
| 71 |
+
| Latin | es, de, it, pt | ✓ | — | 1.3 % each |
|
| 72 |
+
| Cyrillic | ru | ✓ | — | 1.3 % |
|
| 73 |
+
|
| 74 |
+
The rule behind those numbers: **a script that inherits nothing from Latin needs raw text**; a Latin
|
| 75 |
+
script only needs tagged prompts, because the alphabet is already covered and roughly 250 tags are
|
| 76 |
+
enough to attach a language to it. That is why Chinese, Korean and French carry raw lots and Spanish
|
| 77 |
+
does not.
|
| 78 |
+
|
| 79 |
+
**The one genuinely new lot is raw Arabic** — 550 000 characters. Arabic was the last non-Latin script
|
| 80 |
+
still living on tagged prompts alone.
|
| 81 |
+
|
| 82 |
+
The training itself was also restarted from scratch rather than topped up. A network keeps the order it
|
| 83 |
+
learned in: whatever comes last weighs more, and lowering the learning rate on a top-up run does not
|
| 84 |
+
remove that imbalance, it only arbitrates between preserving what was acquired and correcting it. Same
|
| 85 |
+
architecture and same hyper-parameters as v3 — `hidden 32768`, `depth 1`, `tap 24`, `lr 1e-3`, no linear
|
| 86 |
+
path — so what the benchmark below compares is the corpus, not the recipe.
|
| 87 |
+
|
| 88 |
+
### What it buys
|
| 89 |
+
|
| 90 |
+
Phoneme errors against the 32B, averaged over the four files of each generation, the three seeds and
|
| 91 |
+
compared against the threshold:
|
| 92 |
+
|
| 93 |
+
| | threshold | v3 | **v3.1** | |
|
| 94 |
+
|---|---|---|---|---|
|
| 95 |
+
| es | 0.0 | 1.9 | **0.5** | −74 % |
|
| 96 |
+
| de | 2.7 | 9.8 | **3.6** | −64 % |
|
| 97 |
+
| fr | 4.0 | 8.2 | **3.0** | −64 % |
|
| 98 |
+
| it | 2.3 | 4.2 | **1.6** | −63 % |
|
| 99 |
+
| ru | 10.7 | 21.5 | **13.6** | −37 % |
|
| 100 |
+
| ar | 6.3 | 11.1 | **7.6** | −32 % |
|
| 101 |
+
| zh | 3.3 | 3.0 | **2.2** | −25 % |
|
| 102 |
+
| ja | 6.7 | 7.5 | **6.4** | −14 % |
|
| 103 |
+
| ko | 11.3 | 12.8 | **11.5** | −10 % |
|
| 104 |
+
| pt | 17.0 | 25.1 | **24.1** | −4 % |
|
| 105 |
+
| en | 0.0 | **0.2** | 0.5 | +0.3 |
|
| 106 |
+
| **mean** | 5.8 | **9.6** | **6.8** | **−29 %** |
|
| 107 |
+
|
| 108 |
+
The European languages, which v3 only ever saw as a top-up, gain 60 to 74 %. Arabic gains 32 %, which is
|
| 109 |
+
where the new raw-text lot shows up. Russian gains 37 %.
|
| 110 |
+
|
| 111 |
+
**Raw text does not predict the outcome.** Rapported to each language's own threshold, the two groups
|
| 112 |
+
overlap completely:
|
| 113 |
+
|
| 114 |
+
| | script | raw text | threshold | v3 | v3.1 | × threshold |
|
| 115 |
+
|---|---|---|---|---|---|---|
|
| 116 |
+
| zh | Han | ✓ | 3.3 | 3.0 | 2.2 | **0.67** |
|
| 117 |
+
| it | Latin | — | 2.3 | 4.2 | 1.6 | **0.68** |
|
| 118 |
+
| fr | Latin | ✓ | 4.0 | 8.2 | 3.0 | 0.75 |
|
| 119 |
+
| ja | Kana/Kanji | — | 6.7 | 7.5 | 6.4 | 0.96 |
|
| 120 |
+
| ko | Hangul | ✓ | 11.3 | 12.8 | 11.5 | 1.01 |
|
| 121 |
+
| ar | Arabic | ✓ | 6.3 | 11.1 | 7.6 | 1.20 |
|
| 122 |
+
| ru | Cyrillic | — | 10.7 | 21.5 | 13.6 | 1.27 |
|
| 123 |
+
| de | Latin | — | 2.7 | 9.8 | 3.6 | 1.34 |
|
| 124 |
+
| pt | Latin | — | 17.0 | 25.1 | 24.1 | 1.42 |
|
| 125 |
+
|
| 126 |
+
Languages with a raw lot run 0.67 to 1.20; languages without run 0.68 to 1.42. Italian, with tagged
|
| 127 |
+
prompts only, lands second best overall.
|
| 128 |
+
|
| 129 |
+
**Russian is the clearest case.** It is the only non-Latin script here with no raw text and nothing to
|
| 130 |
+
inherit — Japanese borrows kanji from the Chinese lots, Latin scripts borrow the alphabet from English —
|
| 131 |
+
and it still gains **37 %** between v3 and v3.1 on tagged prompts alone, finishing ahead of German and
|
| 132 |
+
Portuguese, which are Latin. The rule written in the build scripts — *250 tags are enough once the
|
| 133 |
+
alphabet is covered* — evidently extends to Cyrillic, which the Qwen3-VL tokenizer covers natively.
|
| 134 |
+
|
| 135 |
+
So raw text is what an **unseen script** needs, not what a language needs. Where a script is already in
|
| 136 |
+
the tokenizer's reach, tags carry it.
|
| 137 |
+
|
| 138 |
+
**English pays for it, and the bill is half a phoneme out of 77.** That is the whole cost of rebalancing:
|
| 139 |
+
v3 was 0.2 errors, v3.1 is 0.5, both far below anything audible and below what a single seed resolves.
|
| 140 |
+
Portuguese barely moves, but nothing moves in Portuguese — the reference itself scatters by 17 there.
|
| 141 |
+
|
| 142 |
+
The net effect is a change of category rather than a better score. v3 sits at **1.46 to 1.77 times** the
|
| 143 |
+
threshold; v3.1 sits at **1.09 to 1.20**. From measurably worse than a seed change, to indistinguishable
|
| 144 |
+
from one.
|
| 145 |
+
|
| 146 |
+
---
|
| 147 |
+
|
| 148 |
+
## The problem with every number I published before
|
| 149 |
+
|
| 150 |
+
A cosine of 0.79, or "23 character errors out of 869" — neither has a scale. Is 23 good? Compared to what? Zero errors is not the right target either, because **the 32B does not reproduce itself**. Change nothing but the seed and it re-pronounces the sentence differently.
|
| 151 |
+
|
| 152 |
+
So the reference is not perfection. It is the 32B compared to itself, same prompt, different seed:
|
| 153 |
+
|
| 154 |
+
| | 32B against itself |
|
| 155 |
+
|---|---|
|
| 156 |
+
| Speech | **5.8 phonemes out of 75** (7.8 %) |
|
| 157 |
+
| Image | **0.9552** SigLIP2 cosine (floor: 0.5313) |
|
| 158 |
+
|
| 159 |
+
That gap is the unit. Everything below is normalised so that **32B = 100**:
|
| 160 |
+
|
| 161 |
+
- **100** — swapping the encoder moves the output as much as changing the seed
|
| 162 |
+
- **above 100** — it moves it less
|
| 163 |
+
- **below 100** — it moves it more
|
| 164 |
+
|
| 165 |
+
Below that threshold you are no longer measuring the projection. You are measuring the generator.
|
| 166 |
+
|
| 167 |
+
---
|
| 168 |
+
|
| 169 |
+
## Files
|
| 170 |
+
|
| 171 |
+
Put them in `ComfyUI/models/clip_projections/`.
|
| 172 |
+
|
| 173 |
+
| File | Encoder | Head | Size | Encoder + projection |
|
| 174 |
+
|---|---|---|---|---|
|
| 175 |
+
| `mmh3-4b-ClipProj-v3.1` | any Qwen3-VL-4B | ridge | 26 MB | **4.6 GB** |
|
| 176 |
+
| `mmh3-4b-ClipProj-v3.1-mlp` | any Qwen3-VL-4B | ridge + residual | 481 MB | 5.1 GB |
|
| 177 |
+
| `mmh3-8b-ClipProj-v3.1` | any Qwen3-VL-8B | ridge | 41 MB | 9.6 GB |
|
| 178 |
+
| `mmh3-8b-ClipProj-v3.1-mlp` | any Qwen3-VL-8B | ridge + residual | 577 MB | 10.1 GB |
|
| 179 |
+
|
| 180 |
+
The 8B matrices expect 4096 input dimensions instead of 2560; the node checks the width and refuses a mismatch.
|
| 181 |
+
|
| 182 |
+
---
|
| 183 |
+
|
| 184 |
+
## The benchmark
|
| 185 |
+
|
| 186 |
+
Nothing here is a single render. Every figure comes from **three seeds — 42, 100 000 and 100 000 000** — chosen far apart so no one can suspect they are correlated.
|
| 187 |
+
|
| 188 |
+
| | volume |
|
| 189 |
+
|---|---|
|
| 190 |
+
| Speech | 297 renders — 9 conditionings × 11 languages × 3 seeds |
|
| 191 |
+
| Image | 405 renders — 9 conditionings × 15 prompts × 3 seeds |
|
| 192 |
+
|
| 193 |
+
**Speech** is scored in phonemes, by [ZIPA-CR-large](https://huggingface.co/anyspeech/zipa-large-crctc-ns-800k) (88 languages, no lexical decoder — it will not silently repair a botched syllable into a real word), against the 32B **of the same seed**. Distances are Levenshtein throughout.
|
| 194 |
+
|
| 195 |
+
**Image** is scored by SigLIP2 so400m, both against the 32B's render and against the prompt itself — on that second axis the 32B is just one column among nine.
|
| 196 |
+
|
| 197 |
+
Languages: en, fr, es, de, it, pt, ru, ar, zh, ja, ko.
|
| 198 |
+
|
| 199 |
+
---
|
| 200 |
+
|
| 201 |
+
## Results
|
| 202 |
+
|
| 203 |
+
| Conditioning | Speech | ± | Image | Prompt |
|
| 204 |
+
|---|---|---|---|---|
|
| 205 |
+
| **32B** *(reference)* | **100.0** | — | **100.0** | **100.0** |
|
| 206 |
+
| `8b-ClipProj-v3.1` | **98.8** | ±1.4 | 100.0 | 100.4 |
|
| 207 |
+
| `8b-ClipProj-v3.1-mlp` | 98.2 | ±1.2 | 101.4 | **102.3** |
|
| 208 |
+
| `4b-ClipProj-v3.1` | 97.9 | ±1.0 | 100.1 | 99.4 |
|
| 209 |
+
| `4b-ClipProj-v3.1-mlp` | 97.8 | ±1.1 | 100.9 | 100.3 |
|
| 210 |
+
| *v3 ridge / mlp, 4B and 8B* | *93.2 – 95.6* | | *99.4 – 102.4* | *99.0 – 100.7* |
|
| 211 |
+
|
| 212 |
+
`±` is the spread of the score across the three seeds. **Two models separated by less than that are not separated at all.**
|
| 213 |
+
|
| 214 |
+
### The raw counts behind the speech score
|
| 215 |
+
|
| 216 |
+
Three metrics, three units, never added together. **PER** counts phonemes over three seeds on the current
|
| 217 |
+
protocol; **WER** and **CER** count words and characters as Whisper hears them, single-seed on the earlier
|
| 218 |
+
0.3 MP protocol. They are listed side by side because they disagree in useful ways — see the language
|
| 219 |
+
breakdown below.
|
| 220 |
+
|
| 221 |
+
| | **PER** (3 seeds) | | **WER** (1 seed) | **CER** (1 seed) |
|
| 222 |
+
|---|---|---|---|---|
|
| 223 |
+
| **32B** *(reference)* | **0 / 2469** | — | 6 / 174 | 6 / 869 |
|
| 224 |
+
| *32B against itself* | *~193 / 2469* | *7.8 %* | — | — |
|
| 225 |
+
| `8b-v3.1` | **211 / 2469** | 8.5 % | **10 / 174** | **14 / 869** |
|
| 226 |
+
| `8b-v3.1-mlp` | 222 / 2469 | 9.0 % | 15 / 174 | 28 / 869 |
|
| 227 |
+
| `4b-v3.1-mlp` | 230 / 2469 | 9.3 % | 13 / 174 | 23 / 869 |
|
| 228 |
+
| `4b-v3.1` | 232 / 2469 | 9.4 % | 15 / 174 | 30 / 869 |
|
| 229 |
+
| `4b-v3-mlp` | 281 / 2469 | 11.4 % | 18 / 174 | 32 / 869 |
|
| 230 |
+
| `8b-v3-mlp` | 317 / 2469 | 12.8 % | 17 / 174 | 29 / 869 |
|
| 231 |
+
| `8b-v3` | 326 / 2469 | 13.2 % | 23 / 174 | 46 / 869 |
|
| 232 |
+
| `4b-v3` | 341 / 2469 | 13.8 % | 19 / 174 | 36 / 869 |
|
| 233 |
+
|
| 234 |
+
The 32B scores 0 on PER by construction — it *is* the reference. The row below it is the meaningful one:
|
| 235 |
+
compared to **itself** on another seed it drifts by about 7.8 %, and the four v3.1 files sit at 8.5 to
|
| 236 |
+
9.4 %. The v3 files sit at 11.4 to 13.8 %, clear of that band.
|
| 237 |
+
|
| 238 |
+
WER and CER rank the files in nearly the same order, which is the point of quoting both: `8b-v3.1` leads
|
| 239 |
+
all three metrics, and no v3 file beats any v3.1 file on any of them.
|
| 240 |
+
|
| 241 |
+
### Why some scores exceed 100 — and why that is not "better than the 32B"
|
| 242 |
+
|
| 243 |
+
The two image columns do not share a reference, and neither exceedance means what it looks like.
|
| 244 |
+
|
| 245 |
+
**Prompt.** This axis is `cos(image embedding, prompt embedding)`. The 32B is **not** the reference here —
|
| 246 |
+
it is one column among nine, and its value is set to 100 only to give the scale a fixed point. Nothing
|
| 247 |
+
requires it to be the best, and it demonstrably is not: on the prompt asking for a loaf **cut in two**, it
|
| 248 |
+
renders a single piece on two seeds out of three. A projection that follows the description more closely
|
| 249 |
+
earns a higher cosine, legitimately.
|
| 250 |
+
|
| 251 |
+
**Image.** Here 100 *is* the 32B against itself, but the comparison is asymmetric: the threshold pits
|
| 252 |
+
`32B(seed 42)` against `32B(seed 7391)` — two different draws — while a projection is compared to
|
| 253 |
+
`32B(seed 42)`, the **same** draw. It plays with its reference's seed, so the draw noise is removed on its
|
| 254 |
+
side. A score of 101.4 says only *closer to that 32B render than two 32B renders are to each other*.
|
| 255 |
+
|
| 256 |
+
**And none of it is significant.** The nine models span 0.0048 of cosine on the prompt axis, against a
|
| 257 |
+
within-model standard deviation of 0.024 to 0.029 — five times larger. The paired test over 45 cases calls
|
| 258 |
+
all eight projections indistinguishable from the 32B, including the one at 102.3. The +2.3 % is real as a
|
| 259 |
+
measurement and void as a result.
|
| 260 |
+
|
| 261 |
+
### What actually separates
|
| 262 |
+
|
| 263 |
+
**The corpus, not the size and not the head.** All four v3.1 land within one point of each other — 97.8 to 98.8 — while ranging from 4.6 to 10.1 GB. All four v3 sit a clear notch below, 93.2 to 95.6, at identical sizes. A 4B v3.1 beats an 8B v3 by four points while weighing half as much.
|
| 264 |
+
|
| 265 |
+
**Nothing separates in image.** All nine conditionings, v3 included, are at or above the threshold: 99.4 to 102.4. Swapping the 32B for a 4B changes the picture **less than changing the seed does**. On this axis the 32B is not a ceiling — `8b-v3.1-mlp` scores 102.3 for prompt fidelity, and on one prompt asking for a loaf cut in two, the 32B rendered a single piece on two seeds out of three while the 8B ridge rendered two on all three.
|
| 266 |
+
|
| 267 |
+
**4B against 8B does not separate on general pronunciation.** 97.8 against 98.8, for a seed-to-seed spread of ±1.0 to ±1.4. If you need one number: they are the same.
|
| 268 |
+
|
| 269 |
+
**Ridge against MLP does not separate either.** The residual buys nothing measurable in speech. It shows up in image prompt fidelity — 102.3 against 100.4 on the 8B — but that axis has its own noise and I would not choose a file on it.
|
| 270 |
+
|
| 271 |
+
### The image measurements in full
|
| 272 |
+
|
| 273 |
+
Two independent SigLIP2 so400m readings over the same 405 renders. First, resemblance to the 32B's own render, same prompt and same seed — 45 cases per model:
|
| 274 |
+
|
| 275 |
+
| | cosine | std. dev. | % of threshold | worst prompt |
|
| 276 |
+
|---|---|---|---|---|
|
| 277 |
+
| **32B against itself** | **0.9552** | — | **100.0** | — |
|
| 278 |
+
| `8b-v3-mlp` | 0.9653 | 0.0328 | 102.4 | 0.8804 |
|
| 279 |
+
| `8b-v3.1-mlp` | 0.9612 | 0.0363 | 101.4 | 0.8808 |
|
| 280 |
+
| `8b-v3` | 0.9594 | 0.0350 | 101.0 | 0.8910 |
|
| 281 |
+
| `4b-v3.1-mlp` | 0.9590 | 0.0321 | 100.9 | 0.9038 |
|
| 282 |
+
| `4b-v3-mlp` | 0.9588 | 0.0408 | 100.8 | 0.8989 |
|
| 283 |
+
| `4b-v3.1` | 0.9557 | 0.0445 | 100.1 | 0.8766 |
|
| 284 |
+
| `8b-v3.1` | 0.9552 | 0.0448 | 100.0 | 0.8893 |
|
| 285 |
+
| `4b-v3` | 0.9528 | 0.0423 | 99.4 | 0.8861 |
|
| 286 |
+
|
| 287 |
+
*Floor: 0.5313 — two 32B renders sharing no content at all still score that, on style and generator artefacts alone.*
|
| 288 |
+
|
| 289 |
+
**The standard deviation settles it.** It runs 0.032 to 0.045, while the entire spread from best to worst model is 0.0125. The scatter within one model is three to four times the gap between models. Nothing here is a ranking.
|
| 290 |
+
|
| 291 |
+
Second, fidelity to the written prompt — an axis where the 32B is one column among nine rather than the reference:
|
| 292 |
+
|
| 293 |
+
| | cosine | std. dev. | base 100 |
|
| 294 |
+
|---|---|---|---|
|
| 295 |
+
| `8b-v3.1-mlp` | 0.1504 | 0.0252 | **102.3** |
|
| 296 |
+
| `4b-v3-mlp` | 0.1481 | 0.0275 | 100.7 |
|
| 297 |
+
| `8b-v3.1` | 0.1477 | 0.0290 | 100.4 |
|
| 298 |
+
| `4b-v3.1-mlp` | 0.1475 | 0.0249 | 100.3 |
|
| 299 |
+
| **32B** | 0.1471 | 0.0245 | **100.0** |
|
| 300 |
+
| `4b-v3.1` | 0.1462 | 0.0272 | 99.4 |
|
| 301 |
+
| `8b-v3` | 0.1460 | 0.0267 | 99.3 |
|
| 302 |
+
| `8b-v3-mlp` | 0.1458 | 0.0236 | 99.2 |
|
| 303 |
+
| `4b-v3` | 0.1456 | 0.0268 | 99.0 |
|
| 304 |
+
|
| 305 |
+
*Floor: −0.0293 — one scene's image against another scene's prompt.*
|
| 306 |
+
|
| 307 |
+
Same verdict, and harder: the spread across all nine models is 0.0048 for a standard deviation of 0.024 to 0.029, **five times larger**. Four models sit above the 32B and four below, in an order that carries no information.
|
| 308 |
+
|
| 309 |
+
The two image axes do not even agree with each other: `8b-v3-mlp` tops the resemblance table and sits second from last on prompt fidelity. Imitating the 32B and following the prompt are not the same objective — the 32B itself misses prompts.
|
| 310 |
+
|
| 311 |
+
---
|
| 312 |
+
|
| 313 |
+
## What counts as an error
|
| 314 |
+
|
| 315 |
+
One error is **one phoneme inserted, deleted or substituted** relative to what the 32B pronounced — same prompt, same seed. Levenshtein distance, nothing weighted, nothing forgiven.
|
| 316 |
+
|
| 317 |
+
There is no dictionary in the loop. ZIPA transcribes sound to IPA and has no lexical decoder, so it will not quietly repair a botched syllable into a real word the way a speech-to-text engine would. What it writes down is what came out of the speaker.
|
| 318 |
+
|
| 319 |
+
Concretely, on the French line *"la lumière de Marseille"*, seed 42 — the phonemes following `d ɛ` ("de"):
|
| 320 |
+
|
| 321 |
+
| | | heard as | errors on the line |
|
| 322 |
+
|---|---|---|---|
|
| 323 |
+
| **32B** | `m a ʀ s ɛ j` | *Marseille* | — |
|
| 324 |
+
| `8b-v3.1` | `m a ʀ s ɛ j` | *Marseille* | 4 / 71 |
|
| 325 |
+
| `4b-v3.1` | `m a ʀ s ɛ ʀ ɛ` | *"marcerre"* | 5 / 71 |
|
| 326 |
+
| `4b-v3` | `m a z ɛ ʀ` | *"mazer"* | 8 / 71 |
|
| 327 |
+
|
| 328 |
+
Note how little the toponym costs: `4b-v3.1` botches the name outright and pays **one** phoneme more than `8b-v3.1` over the whole sentence. That is exactly why the aggregate scores cannot settle the proper-noun question, and why it gets its own section below rather than a place in the ranking.
|
| 329 |
+
|
| 330 |
+
## Per language, because the average hides everything
|
| 331 |
+
|
| 332 |
+
Errors are counted against the 32B **of the same seed**, averaged over the three seeds. The first two columns are the yardstick: how long the reference is, and how much the 32B differs from *itself*.
|
| 333 |
+
|
| 334 |
+
| | length | **threshold** | `8b-v3.1` | `8b-v3.1-mlp` | `4b-v3.1` | `4b-v3.1-mlp` |
|
| 335 |
+
|---|---|---|---|---|---|---|
|
| 336 |
+
| en | 77 | **0.0** | 0.7 | 0.7 | 0.7 | 0.0 |
|
| 337 |
+
| es | 62 | **0.0** | 1.3 | 0.0 | 0.7 | 0.0 |
|
| 338 |
+
| de | 97 | 2.7 | 3.0 | 5.0 | 3.7 | 2.7 |
|
| 339 |
+
| it | 62 | 2.3 | 0.7 | 0.3 | 4.3 | 1.0 |
|
| 340 |
+
| zh | 85 | 3.3 | 3.3 | 1.0 | 2.7 | 2.0 |
|
| 341 |
+
| fr | 70 | 4.0 | 1.7 | 1.3 | 5.0 | 4.0 |
|
| 342 |
+
| ar | 94 | 6.3 | 5.7 | 4.7 | 7.7 | 12.3 |
|
| 343 |
+
| ja | 73 | 6.7 | 6.3 | 6.7 | 5.7 | 7.0 |
|
| 344 |
+
| ru | 77 | 10.7 | 14.0 | 16.3 | 14.0 | 10.0 |
|
| 345 |
+
| ko | 64 | 11.3 | 11.0 | 15.0 | 10.7 | 9.3 |
|
| 346 |
+
| pt | 62 | **17.0** | 22.7 | 23.0 | 22.3 | 28.3 |
|
| 347 |
+
|
| 348 |
+
Read it against the threshold column, never in absolute terms:
|
| 349 |
+
|
| 350 |
+
- **English and Spanish** — the 32B repeats itself phoneme for phoneme. There, a single phoneme of drift is real signal, and all four files stay within one.
|
| 351 |
+
- **French, Italian, Chinese, Arabic, Japanese** — the projections are *at or below* the 32B's own variance. In French both 8B files land at 1.7 and 1.3 against a threshold of 4.0: closer to the 32B than the 32B is to itself.
|
| 352 |
+
- **Portuguese, Russian and Korean** carry thresholds of 17.0, 10.7 and 11.3 — the reference rewrites a large share of its own pronunciation between seeds. Any single-seed comparison there was measuring the dice.
|
| 353 |
+
|
| 354 |
+
### Where the phoneme metric misleads, and the cross-check that catches it
|
| 355 |
+
|
| 356 |
+
A high threshold does not mean the speech is bad. It means **the phoneme transcriber cannot hold that
|
| 357 |
+
language still.** Cross-checking against Whisper, which reads words rather than sounds, on the same
|
| 358 |
+
renders:
|
| 359 |
+
|
| 360 |
+
| | ZIPA threshold | ZIPA v3.1 | × threshold | **Whisper, 32B** | **Whisper, v3.1** |
|
| 361 |
+
|---|---|---|---|---|---|
|
| 362 |
+
| pt | 17.0 | 24.1 | **1.42** | **0 / 88** | **1.8 / 88** |
|
| 363 |
+
| ru | 10.7 | 13.6 | 1.27 | **0 / 85** | 2.8 / 85 |
|
| 364 |
+
| ko | 11.3 | 11.5 | 1.01 | 2 / 85 | 3.2 / 85 |
|
| 365 |
+
| ja | 6.7 | 6.4 | 0.96 | 2 / 39 | 5.2 / 39 |
|
| 366 |
+
| zh | 3.3 | 2.2 | 0.67 | 2 / 31 | 3.5 / 31 |
|
| 367 |
+
|
| 368 |
+
**Portuguese is the worst language by phoneme and one of the best by word** — zero character errors for
|
| 369 |
+
the 32B, 2 % for the v3.1 files. Russian likewise: Whisper transcribes the 32B and two of the projections
|
| 370 |
+
word for word.
|
| 371 |
+
|
| 372 |
+
The cause is exactly what makes ZIPA useful elsewhere: it has no lexical decoder. European Portuguese
|
| 373 |
+
elides and reduces its vowels, Russian has vowel reduction under stress shift — the phonetic realisation
|
| 374 |
+
moves from one draw to the next while the word does not. ZIPA counts every allophonic variation as an
|
| 375 |
+
error; Whisper, which recognises the word, sees none. In Japanese and Chinese the bias runs the other
|
| 376 |
+
way: Whisper is harsher, because one missed ideogram weighs heavily on 31 characters.
|
| 377 |
+
|
| 378 |
+
**Neither metric is sufficient alone.** Where the ZIPA threshold is high, read the word column.
|
| 379 |
+
(Whisper figures are single-seed, on the earlier 0.3 MP protocol.)
|
| 380 |
+
|
| 381 |
+
---
|
| 382 |
+
|
| 383 |
+
## The 32B is one of the least stable models here
|
| 384 |
+
|
| 385 |
+
Distance between two renders of the **same** model, seed changed, nothing else:
|
| 386 |
+
|
| 387 |
+
| | against itself | against the 32B |
|
| 388 |
+
|---|---|---|
|
| 389 |
+
| `4b-v3.1` | **3.6** | 7.0 |
|
| 390 |
+
| `4b-v3.1-mlp` | 4.0 | 7.0 |
|
| 391 |
+
| `8b-v3.1` | 4.1 | 6.4 |
|
| 392 |
+
| `8b-v3.1-mlp` | 4.7 | 6.7 |
|
| 393 |
+
| **32B** | **5.8** | — |
|
| 394 |
+
|
| 395 |
+
The projections repeat themselves *better* than the model they imitate.
|
| 396 |
+
|
| 397 |
+
**And the gap to the 32B is reproducible, not random.** Each projection sits far closer to itself (3.6–4.7) than to the 32B (6.4–7.0). If swapping the encoder merely added randomness, those two columns would match. They do not — each file redoes the same offset on every seed.
|
| 398 |
+
|
| 399 |
+
**It is not an accent either.** An accent would mean one phoneme consistently rendered as another. Counting the actual substitutions says otherwise:
|
| 400 |
+
|
| 401 |
+
| | substitutions | covered by recurring patterns |
|
| 402 |
+
|---|---|---|
|
| 403 |
+
| **32B against itself** | **103** | ɑ→a ×10, ɾ→r ×6, ʒ→ʐ ×5 |
|
| 404 |
+
| `8b-v3.1` | **103** | 4 % — one pattern |
|
| 405 |
+
| `4b-v3.1` | 114 | **0 %** |
|
| 406 |
+
| `8b-v3.1-mlp` | 115 | 8 % |
|
| 407 |
+
| `4b-v3.1-mlp` | 125 | **0 %** |
|
| 408 |
+
| the four v3 files | 153–196 | 2–13 % |
|
| 409 |
+
|
| 410 |
+
The v3.1 files produce **as many substitutions as the 32B inflicts on itself** — 103 to 125 against 103 — and almost none of them form a repeating pattern. The offset is reproducible but scattered across many different sounds rather than concentrated into a signature. Ironically the clearest patterns belong to the 32B itself, between its own seeds, where they are ordinary allophonic variation.
|
| 411 |
+
|
| 412 |
+
The v3 files produce 1.5 to 2 times as many.
|
| 413 |
+
|
| 414 |
+
None of which tells you what it *sounds* like. A native speaker might well hear something none of these counts describe.
|
| 415 |
+
|
| 416 |
+
---
|
| 417 |
+
|
| 418 |
+
## One word, one language
|
| 419 |
+
|
| 420 |
+
The French prompt says *"la lumière de **Marseille**"*. Six phonemes out of seventy — the error rate drowns them, the ear does not:
|
| 421 |
+
|
| 422 |
+
| | seed 42 | seed 100 k | seed 100 M |
|
| 423 |
+
|---|---|---|---|
|
| 424 |
+
| 32B | ✓ | ✓ | ✓ |
|
| 425 |
+
| `8b-v3.1` | ✓ | ✓ | ✓ |
|
| 426 |
+
| `8b-v3.1-mlp` | ✓ | ✓ | ✓ |
|
| 427 |
+
| `4b-v3.1-mlp` | ✗ | ✗ | ✓ |
|
| 428 |
+
| `4b-v3.1` | ✗ | ✗ | ✗ |
|
| 429 |
+
|
| 430 |
+
Treat this for what it is: **one proper noun, in one of eleven languages, on three seeds.** It is not a ranking criterion and the aggregate scores already say 4B and 8B are equivalent. It is a hint that the two sizes may diverge on rare lexical items even where they agree on everything else — consistent with quantisation costing facts, which was already known. If your prompts lean on proper nouns, test both before deciding.
|
| 431 |
+
|
| 432 |
+
---
|
| 433 |
+
|
| 434 |
+
## Which one to take
|
| 435 |
+
|
| 436 |
+
| If you | Take |
|
| 437 |
+
|---|---|
|
| 438 |
+
| generate images or video without speech | **`4b-ClipProj-v3.1`** — 4.6 GB, indistinguishable from the 32B |
|
| 439 |
+
| are tight on VRAM | **`4b-ClipProj-v3.1`** — the ridge is 26 MB and gives up nothing measurable |
|
| 440 |
+
| generate multilingual speech | **`8b-ClipProj-v3.1`** — best speech score, and steady on proper nouns |
|
| 441 |
+
| rely on named people or places | **`8b-ClipProj-v3.1`**, and run the 32B once to check the name works there at all |
|
| 442 |
+
|
| 443 |
+
Do not take a v3: it is the only difference this benchmark resolves cleanly.
|
| 444 |
+
|
| 445 |
+
---
|
| 446 |
+
|
| 447 |
+
## How the speech benchmark got affordable
|
| 448 |
+
|
| 449 |
+
The old protocol rendered full 0.3 MP video and threw the picture away. Measured, same prompt and seed:
|
| 450 |
+
|
| 451 |
+
| | time |
|
| 452 |
+
|---|---|
|
| 453 |
+
| 0.3 MP video + audio + previews *(old)* | 80.3 s |
|
| 454 |
+
| same, video VAE and previews removed | 48.7 s |
|
| 455 |
+
| **64×64, 8 steps, audio only** | **12.4 s** |
|
| 456 |
+
|
| 457 |
+
Verified lossless before adopting: **+0.2 dB** across every octave band and **1 phoneme out of 70** for dropping the video decode; 64×64 costs 3 phonemes out of 70 against the full-resolution render.
|
| 458 |
+
|
| 459 |
+
**128×128 was rejected** — it truncates the start of the sentence, exactly the same eight phonemes at 6 steps and at 8. 64×64 does not. Counter-intuitive, reproducible, and the reason the whole benchmark runs at the smaller size.
|
| 460 |
+
|
| 461 |
+
That is what made three seeds across 297 renders possible at all: 31 minutes on two cards instead of six and a half hours.
|
| 462 |
+
|
| 463 |
+
---
|
| 464 |
+
|
| 465 |
+
## Limitations
|
| 466 |
+
|
| 467 |
+
**Three seeds fix the order of magnitude of the noise, not its tail.** Any gap under one point of score is not a result.
|
| 468 |
+
|
| 469 |
+
**The cosine is blind to countable attributes.** A whole loaf and a halved loaf, same crust, same paper, same light, give the same vector to the fourth decimal. Image equivalence here means *global appearance*, not attribute-by-attribute conformity.
|
| 470 |
+
|
| 471 |
+
**Speech quality is deliberately poor.** Six to eight steps gives a tinny, canned sound — identically for the 32B, with the same 19 dB dip between 1 and 3 kHz. The benchmark measures **correctness of pronunciation, not fidelity of reproduction**.
|
| 472 |
+
|
| 473 |
+
**The phoneme metric is unreliable in Portuguese, Russian and Korean** — not the speech itself. The
|
| 474 |
+
reference drifts by 17.0, 10.7 and 11.3 phonemes there between seeds, while Whisper transcribes the same
|
| 475 |
+
renders with zero to three character errors. Read the word column in those languages.
|
| 476 |
+
|
| 477 |
+
**Quantisation costs facts.** Known before, still true, and the most likely explanation for the proper-noun gap.
|
| 478 |
+
|
| 479 |
+
---
|
| 480 |
+
|
| 481 |
+
## Licence and responsibility
|
| 482 |
+
|
| 483 |
+
MIT, like the node. These matrices are derived from the activations of both models and their legal status is unclear; they are provided as-is, for research.
|
| 484 |
+
|
| 485 |
+
- **Qwen3-VL** — Alibaba, Apache 2.0.
|
| 486 |
+
- **MiniMax H3** — custom licence, read it before any commercial use.
|
| 487 |
+
|
| 488 |
+
Not affiliated with, endorsed by, or connected to Alibaba / Qwen, MiniMax, or Comfy Org. You remain responsible for what you generate.
|
| 489 |
+
|
| 490 |
+
---
|
| 491 |
+
|
| 492 |
+
## Credits
|
| 493 |
+
|
| 494 |
+
Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number here was measured on this hardware, never estimated. Where a prediction lost to a measurement, the measurement won and the text was rewritten — which happened three times in this release, the largest being a single-seed ranking of the v3.1 files that dissolved entirely once the threshold was known.
|
bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e53d45ea1012723b3857fb14471a40455538254236edac1b437bf017d40fdffd
|
| 3 |
+
size 429755
|
bench3.1/audio/ru/4b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:14b85ea705b93d1e4c0410682699b17bce06c6a0bd32415ee3515e86f88a67b8
|
| 3 |
+
size 453961
|
bench3.1/audio/ru/4b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:29607b7fa47691dbbc453ce9122db5978aaa8b4263f4bf757e2df21f915e5f7d
|
| 3 |
+
size 458935
|
bench3.1/audio/ru/4b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:10b17ba27faed991a156909765e22fb28d79d838bc2f01f8c1f90bd5c5c358ca
|
| 3 |
+
size 424544
|
bench3.1/audio/ru/4b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b915e71b290af104f38fd9123e4642c82673b1f59c09cc156462bb847a25515c
|
| 3 |
+
size 447572
|
bench3.1/audio/ru/4b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b666c951a80e641a69dd57a7c13c0f64a6638c8b108b47e7ef3e6029ee5c38dc
|
| 3 |
+
size 444602
|
bench3.1/audio/ru/4b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fea28c5ba8d35e83900d7849288250acd27bfe3e2e53578e46e2012421102fc1
|
| 3 |
+
size 423455
|
bench3.1/audio/ru/4b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:96eb417bd32b0b5e41e1df444bef5eb703894705df4271f7582b0aaa5c36388e
|
| 3 |
+
size 456929
|
bench3.1/audio/ru/8b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:70616fe97638cadc8c565e8b9be5d6f19eb12b14f716c71d0f4d7076f3848694
|
| 3 |
+
size 470866
|
bench3.1/audio/ru/8b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0ce3fed11e203360f22629ec594545bf49fc358353649a7c9355c722e7c80fd0
|
| 3 |
+
size 430723
|
bench3.1/audio/ru/8b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:09c5943d5941fad3be131442fa49686347df70a1c8cdefbdf605f6897a8ae8ec
|
| 3 |
+
size 460268
|
bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3bb419c0f71433d86a0ebd5acd9ca657c62bdcea415c41a196c53eab887136fb
|
| 3 |
+
size 468965
|
bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:82c5603addbf88dad2286e88ab9934b21d7c1d1a6e2749f30b1c5692ce24b198
|
| 3 |
+
size 435985
|
bench3.1/audio/ru/8b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:45307a209ab81d094164ca22ae0918b08ae5e7c7ccd0d0783273c814786a68b2
|
| 3 |
+
size 467116
|
bench3.1/audio/ru/8b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e2f5e8d57020673cb2a886dc91ba798020d5da4dcc15ebcd6660ebb8f1870b4c
|
| 3 |
+
size 468444
|
bench3.1/audio/ru/8b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a6257d388f6c2e52877175e2772690649683229daddfcd9181237b0e2905b4f9
|
| 3 |
+
size 435118
|
bench3.1/audio/ru/8b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a42b305503a82ae3a99f41a1af9d8310ef2ee69fe4c940c0a7b57e80b6f3b306
|
| 3 |
+
size 451851
|
bench3.1/audio/ru/8b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7c518a115cfbc962d60b7c37be08b265f158413581d7772752e468fc2a6116b1
|
| 3 |
+
size 443896
|
bench3.1/audio/ru/8b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0cc08c210a498207db38c69a7559f68dfbe35ceba9432a8dddaadd936822bd7d
|
| 3 |
+
size 429043
|
bench3.1/audio/ru/8b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b8a0c109adde39f7232df81be57dcfc0121469808e2080e7598ad5c9fe47524c
|
| 3 |
+
size 468524
|
bench3.1/audio/zh/32b_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e599bba7269aa0ff769c08cf9c77ef0bb0bf2ee2496a362999a1a5de7e0f8cf0
|
| 3 |
+
size 461955
|
bench3.1/audio/zh/32b_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e35810382807ae12b022cbad36c09b6cc8a33f91b4e749436db64a8f4157ca82
|
| 3 |
+
size 398314
|
bench3.1/audio/zh/32b_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1b5a53b427844cf39df6ba6081ea9ca360e16c39bb2ed0fea2d3d6391d647f77
|
| 3 |
+
size 427973
|
bench3.1/audio/zh/4b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5d8b8f18e68a84190fed01c48343c85cbf678846293edc9bd436075b83b6ddd4
|
| 3 |
+
size 398882
|
bench3.1/audio/zh/4b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dbf70317d461f64035ad8bf0105a139733a23bd7285034b2d9a11ef0b3fee808
|
| 3 |
+
size 377409
|
bench3.1/audio/zh/4b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:903757ce6ea45aba792380893263e876a3234921c625568d01376503c5beb9d1
|
| 3 |
+
size 371468
|
bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1fc3444c32f70d3eedbf26d2568646fb9db7a3cd5a7e83ed99780b0b86306718
|
| 3 |
+
size 413401
|
bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:54b63f5686a68118ebe145e05f99cb7c6d489ca9d1991bcd9aa805c3cdef8e60
|
| 3 |
+
size 373573
|
bench3.1/audio/zh/4b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ff260b95a903f8de90b8c01c0b9c6ae4dbb60b7b6a9f8890bae01d8e6a202523
|
| 3 |
+
size 382691
|
bench3.1/audio/zh/4b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:76d6d0f72e94c95546165d5f573e157fc6d1a23dba9bd1cccc2d748c4e95eb98
|
| 3 |
+
size 389361
|
bench3.1/audio/zh/4b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4f4110d6ee92c00ef0b480dc971e41b47a9bb6dc6140f8be8b90c1704056f402
|
| 3 |
+
size 375759
|
bench3.1/audio/zh/4b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a7d4d3dfc432e1d83c2ed12dbe8f1ad98e7e33e30a8c643431898db5368e7492
|
| 3 |
+
size 384508
|
bench3.1/audio/zh/4b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c949520895168b02258ef3b15960d39c40630dd77f9a0f5d8454558c1e06d364
|
| 3 |
+
size 378740
|
bench3.1/audio/zh/4b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c2785d72b502af9ce29666f640a6edc3afc09d47acf998c74351104d74ab01b0
|
| 3 |
+
size 379253
|
bench3.1/audio/zh/4b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e524606a0168948770a5dbb457c4aae58f1c26261308b627f789efd39b7d555f
|
| 3 |
+
size 376275
|
bench3.1/audio/zh/8b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1afd8836040c25f5198e5f6e682b2587dedb9cbcf91e7dcf3e1922995975154a
|
| 3 |
+
size 401529
|
bench3.1/audio/zh/8b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ba18aea41c4b90f86b2508c3c27034466dd872c9e77247904fec8d45fb5613a6
|
| 3 |
+
size 364176
|
bench3.1/audio/zh/8b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6eaddc0f3ad47127c8c58e59c1c43b50103d6bb11807a681dcbecd20c4f05cb2
|
| 3 |
+
size 372752
|
bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ebf2e43d6246efb00c0bcb951e796ef32a9cd98db9c5efe27a7bbb993354f4b8
|
| 3 |
+
size 413868
|
bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8bb615863a72acfe1a15daa451c0278374ca40f8c66982a1d081d9ea1b57c59d
|
| 3 |
+
size 393327
|
bench3.1/audio/zh/8b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dc76d98d90a4b637f591317a810d9ad36a5b8b3d78d83a72a07dccf61f659514
|
| 3 |
+
size 387031
|
bench3.1/audio/zh/8b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:12791b0ea5979de46cdebb38c26d0411b6fc54638c4067f2d2a2082d8c6a3042
|
| 3 |
+
size 385676
|
bench3.1/audio/zh/8b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a6bb13e971f054676dd2cded560927fea9053a53dee5fc9f1713796921277485
|
| 3 |
+
size 365478
|
bench3.1/audio/zh/8b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:190bd4685e28a1188e8582ad09b0923ce72b8f9f3b15aa4b6a4f228d94bbc219
|
| 3 |
+
size 372457
|
bench3.1/audio/zh/8b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f89932a5e632580703ac9c3214c964eba4b33d71fb5103934642c7bd46c3bc6d
|
| 3 |
+
size 370772
|
bench3.1/audio/zh/8b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8823243f92f9310e50fa6565b5b7df61d104ce18d178967a8f7a9cde373d8681
|
| 3 |
+
size 360986
|
bench3.1/audio/zh/8b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:229f7ea3267d1568e0f4051d8e56604ebcc48c08d13c52c007b0d905cbe5dcc3
|
| 3 |
+
size 369764
|