NicoLab28 commited on
Commit
388c321
·
verified ·
1 Parent(s): 6c1a124

v3.1 matrices, three-seed benchmark and eleven-language demo (part 2)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +236 -0
  2. README.md +173 -192
  3. bench3.1/README.md +494 -0
  4. bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac +3 -0
  5. bench3.1/audio/ru/4b-v3.1-mlp_s42.flac +3 -0
  6. bench3.1/audio/ru/4b-v3.1_s100000.flac +3 -0
  7. bench3.1/audio/ru/4b-v3.1_s100000000.flac +3 -0
  8. bench3.1/audio/ru/4b-v3.1_s42.flac +3 -0
  9. bench3.1/audio/ru/4b-v3_s100000.flac +3 -0
  10. bench3.1/audio/ru/4b-v3_s100000000.flac +3 -0
  11. bench3.1/audio/ru/4b-v3_s42.flac +3 -0
  12. bench3.1/audio/ru/8b-v3-mlp_s100000.flac +3 -0
  13. bench3.1/audio/ru/8b-v3-mlp_s100000000.flac +3 -0
  14. bench3.1/audio/ru/8b-v3-mlp_s42.flac +3 -0
  15. bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac +3 -0
  16. bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac +3 -0
  17. bench3.1/audio/ru/8b-v3.1-mlp_s42.flac +3 -0
  18. bench3.1/audio/ru/8b-v3.1_s100000.flac +3 -0
  19. bench3.1/audio/ru/8b-v3.1_s100000000.flac +3 -0
  20. bench3.1/audio/ru/8b-v3.1_s42.flac +3 -0
  21. bench3.1/audio/ru/8b-v3_s100000.flac +3 -0
  22. bench3.1/audio/ru/8b-v3_s100000000.flac +3 -0
  23. bench3.1/audio/ru/8b-v3_s42.flac +3 -0
  24. bench3.1/audio/zh/32b_s100000.flac +3 -0
  25. bench3.1/audio/zh/32b_s100000000.flac +3 -0
  26. bench3.1/audio/zh/32b_s42.flac +3 -0
  27. bench3.1/audio/zh/4b-v3-mlp_s100000.flac +3 -0
  28. bench3.1/audio/zh/4b-v3-mlp_s100000000.flac +3 -0
  29. bench3.1/audio/zh/4b-v3-mlp_s42.flac +3 -0
  30. bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac +3 -0
  31. bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac +3 -0
  32. bench3.1/audio/zh/4b-v3.1-mlp_s42.flac +3 -0
  33. bench3.1/audio/zh/4b-v3.1_s100000.flac +3 -0
  34. bench3.1/audio/zh/4b-v3.1_s100000000.flac +3 -0
  35. bench3.1/audio/zh/4b-v3.1_s42.flac +3 -0
  36. bench3.1/audio/zh/4b-v3_s100000.flac +3 -0
  37. bench3.1/audio/zh/4b-v3_s100000000.flac +3 -0
  38. bench3.1/audio/zh/4b-v3_s42.flac +3 -0
  39. bench3.1/audio/zh/8b-v3-mlp_s100000.flac +3 -0
  40. bench3.1/audio/zh/8b-v3-mlp_s100000000.flac +3 -0
  41. bench3.1/audio/zh/8b-v3-mlp_s42.flac +3 -0
  42. bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac +3 -0
  43. bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac +3 -0
  44. bench3.1/audio/zh/8b-v3.1-mlp_s42.flac +3 -0
  45. bench3.1/audio/zh/8b-v3.1_s100000.flac +3 -0
  46. bench3.1/audio/zh/8b-v3.1_s100000000.flac +3 -0
  47. bench3.1/audio/zh/8b-v3.1_s42.flac +3 -0
  48. bench3.1/audio/zh/8b-v3_s100000.flac +3 -0
  49. bench3.1/audio/zh/8b-v3_s100000000.flac +3 -0
  50. bench3.1/audio/zh/8b-v3_s42.flac +3 -0
.gitattributes CHANGED
@@ -289,3 +289,239 @@ bench3.1/audio/ru/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
289
  bench3.1/audio/ru/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
290
  bench3.1/audio/ru/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
291
  bench3.1/audio/ru/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
289
  bench3.1/audio/ru/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
290
  bench3.1/audio/ru/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
291
  bench3.1/audio/ru/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
292
+ bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
293
+ bench3.1/audio/ru/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
294
+ bench3.1/audio/ru/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
295
+ bench3.1/audio/ru/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
296
+ bench3.1/audio/ru/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
297
+ bench3.1/audio/ru/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
298
+ bench3.1/audio/ru/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
299
+ bench3.1/audio/ru/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
300
+ bench3.1/audio/ru/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
301
+ bench3.1/audio/ru/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
302
+ bench3.1/audio/ru/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
303
+ bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
304
+ bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
305
+ bench3.1/audio/ru/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
306
+ bench3.1/audio/ru/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
307
+ bench3.1/audio/ru/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
308
+ bench3.1/audio/ru/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
309
+ bench3.1/audio/ru/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
310
+ bench3.1/audio/ru/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
311
+ bench3.1/audio/ru/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
312
+ bench3.1/audio/zh/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
313
+ bench3.1/audio/zh/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
314
+ bench3.1/audio/zh/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
315
+ bench3.1/audio/zh/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
316
+ bench3.1/audio/zh/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
317
+ bench3.1/audio/zh/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
318
+ bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
319
+ bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
320
+ bench3.1/audio/zh/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
321
+ bench3.1/audio/zh/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
322
+ bench3.1/audio/zh/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
323
+ bench3.1/audio/zh/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
324
+ bench3.1/audio/zh/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
325
+ bench3.1/audio/zh/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
326
+ bench3.1/audio/zh/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
327
+ bench3.1/audio/zh/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
328
+ bench3.1/audio/zh/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
329
+ bench3.1/audio/zh/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
330
+ bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
331
+ bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
332
+ bench3.1/audio/zh/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
333
+ bench3.1/audio/zh/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
334
+ bench3.1/audio/zh/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
335
+ bench3.1/audio/zh/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
336
+ bench3.1/audio/zh/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
337
+ bench3.1/audio/zh/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
338
+ bench3.1/audio/zh/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
339
+ bench3.1/images/brutes/p01_32b.png filter=lfs diff=lfs merge=lfs -text
340
+ bench3.1/images/brutes/p01_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
341
+ bench3.1/images/brutes/p01_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
342
+ bench3.1/images/brutes/p01_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
343
+ bench3.1/images/brutes/p01_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
344
+ bench3.1/images/brutes/p02_32b.png filter=lfs diff=lfs merge=lfs -text
345
+ bench3.1/images/brutes/p02_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
346
+ bench3.1/images/brutes/p02_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
347
+ bench3.1/images/brutes/p02_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
348
+ bench3.1/images/brutes/p02_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
349
+ bench3.1/images/brutes/p03_32b.png filter=lfs diff=lfs merge=lfs -text
350
+ bench3.1/images/brutes/p03_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
351
+ bench3.1/images/brutes/p03_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
352
+ bench3.1/images/brutes/p03_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
353
+ bench3.1/images/brutes/p03_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
354
+ bench3.1/images/brutes/p04_32b.png filter=lfs diff=lfs merge=lfs -text
355
+ bench3.1/images/brutes/p04_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
356
+ bench3.1/images/brutes/p04_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
357
+ bench3.1/images/brutes/p04_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
358
+ bench3.1/images/brutes/p04_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
359
+ bench3.1/images/brutes/p05_32b.png filter=lfs diff=lfs merge=lfs -text
360
+ bench3.1/images/brutes/p05_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
361
+ bench3.1/images/brutes/p05_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
362
+ bench3.1/images/brutes/p05_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
363
+ bench3.1/images/brutes/p05_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
364
+ bench3.1/images/brutes/p06_32b.png filter=lfs diff=lfs merge=lfs -text
365
+ bench3.1/images/brutes/p06_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
366
+ bench3.1/images/brutes/p06_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
367
+ bench3.1/images/brutes/p06_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
368
+ bench3.1/images/brutes/p06_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
369
+ bench3.1/images/brutes/p07_32b.png filter=lfs diff=lfs merge=lfs -text
370
+ bench3.1/images/brutes/p07_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
371
+ bench3.1/images/brutes/p07_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
372
+ bench3.1/images/brutes/p07_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
373
+ bench3.1/images/brutes/p07_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
374
+ bench3.1/images/brutes/p08_32b.png filter=lfs diff=lfs merge=lfs -text
375
+ bench3.1/images/brutes/p08_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
376
+ bench3.1/images/brutes/p08_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
377
+ bench3.1/images/brutes/p08_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
378
+ bench3.1/images/brutes/p08_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
379
+ bench3.1/images/brutes/p09_32b.png filter=lfs diff=lfs merge=lfs -text
380
+ bench3.1/images/brutes/p09_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
381
+ bench3.1/images/brutes/p09_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
382
+ bench3.1/images/brutes/p09_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
383
+ bench3.1/images/brutes/p09_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
384
+ bench3.1/images/brutes/p10_32b.png filter=lfs diff=lfs merge=lfs -text
385
+ bench3.1/images/brutes/p10_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
386
+ bench3.1/images/brutes/p10_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
387
+ bench3.1/images/brutes/p10_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
388
+ bench3.1/images/brutes/p10_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
389
+ bench3.1/images/brutes/p11_32b.png filter=lfs diff=lfs merge=lfs -text
390
+ bench3.1/images/brutes/p11_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
391
+ bench3.1/images/brutes/p11_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
392
+ bench3.1/images/brutes/p11_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
393
+ bench3.1/images/brutes/p11_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
394
+ bench3.1/images/brutes/p12_32b.png filter=lfs diff=lfs merge=lfs -text
395
+ bench3.1/images/brutes/p12_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
396
+ bench3.1/images/brutes/p12_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
397
+ bench3.1/images/brutes/p12_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
398
+ bench3.1/images/brutes/p12_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
399
+ bench3.1/images/brutes/p13_32b.png filter=lfs diff=lfs merge=lfs -text
400
+ bench3.1/images/brutes/p13_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
401
+ bench3.1/images/brutes/p13_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
402
+ bench3.1/images/brutes/p13_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
403
+ bench3.1/images/brutes/p13_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
404
+ bench3.1/images/brutes/p14_32b.png filter=lfs diff=lfs merge=lfs -text
405
+ bench3.1/images/brutes/p14_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
406
+ bench3.1/images/brutes/p14_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
407
+ bench3.1/images/brutes/p14_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
408
+ bench3.1/images/brutes/p14_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
409
+ bench3.1/images/brutes/p15_32b.png filter=lfs diff=lfs merge=lfs -text
410
+ bench3.1/images/brutes/p15_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
411
+ bench3.1/images/brutes/p15_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
412
+ bench3.1/images/brutes/p15_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
413
+ bench3.1/images/brutes/p15_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
414
+ bench3.1/images/planches/p01.jpg filter=lfs diff=lfs merge=lfs -text
415
+ bench3.1/images/planches/p02.jpg filter=lfs diff=lfs merge=lfs -text
416
+ bench3.1/images/planches/p03.jpg filter=lfs diff=lfs merge=lfs -text
417
+ bench3.1/images/planches/p04.jpg filter=lfs diff=lfs merge=lfs -text
418
+ bench3.1/images/planches/p05.jpg filter=lfs diff=lfs merge=lfs -text
419
+ bench3.1/images/planches/p06.jpg filter=lfs diff=lfs merge=lfs -text
420
+ bench3.1/images/planches/p07.jpg filter=lfs diff=lfs merge=lfs -text
421
+ bench3.1/images/planches/p08.jpg filter=lfs diff=lfs merge=lfs -text
422
+ bench3.1/images/planches/p09.jpg filter=lfs diff=lfs merge=lfs -text
423
+ bench3.1/images/planches/p10.jpg filter=lfs diff=lfs merge=lfs -text
424
+ bench3.1/images/planches/p12.jpg filter=lfs diff=lfs merge=lfs -text
425
+ bench3.1/images/planches/p13.jpg filter=lfs diff=lfs merge=lfs -text
426
+ bench3.1/images/planches/p14.jpg filter=lfs diff=lfs merge=lfs -text
427
+ bench3.1/images/planches/p15.jpg filter=lfs diff=lfs merge=lfs -text
428
+ bench3.1/video/clipproj-v3.1-eleven-languages.mp4 filter=lfs diff=lfs merge=lfs -text
429
+ bench3.1/video/langues/ar/32b.mp4 filter=lfs diff=lfs merge=lfs -text
430
+ bench3.1/video/langues/ar/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
431
+ bench3.1/video/langues/ar/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
432
+ bench3.1/video/langues/ar/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
433
+ bench3.1/video/langues/ar/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
434
+ bench3.1/video/langues/ar/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
435
+ bench3.1/video/langues/ar/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
436
+ bench3.1/video/langues/ar/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
437
+ bench3.1/video/langues/ar/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
438
+ bench3.1/video/langues/de/32b.mp4 filter=lfs diff=lfs merge=lfs -text
439
+ bench3.1/video/langues/de/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
440
+ bench3.1/video/langues/de/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
441
+ bench3.1/video/langues/de/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
442
+ bench3.1/video/langues/de/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
443
+ bench3.1/video/langues/de/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
444
+ bench3.1/video/langues/de/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
445
+ bench3.1/video/langues/de/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
446
+ bench3.1/video/langues/de/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
447
+ bench3.1/video/langues/en/32b.mp4 filter=lfs diff=lfs merge=lfs -text
448
+ bench3.1/video/langues/en/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
449
+ bench3.1/video/langues/en/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
450
+ bench3.1/video/langues/en/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
451
+ bench3.1/video/langues/en/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
452
+ bench3.1/video/langues/en/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
453
+ bench3.1/video/langues/en/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
454
+ bench3.1/video/langues/en/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
455
+ bench3.1/video/langues/en/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
456
+ bench3.1/video/langues/es/32b.mp4 filter=lfs diff=lfs merge=lfs -text
457
+ bench3.1/video/langues/es/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
458
+ bench3.1/video/langues/es/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
459
+ bench3.1/video/langues/es/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
460
+ bench3.1/video/langues/es/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
461
+ bench3.1/video/langues/es/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
462
+ bench3.1/video/langues/es/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
463
+ bench3.1/video/langues/es/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
464
+ bench3.1/video/langues/es/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
465
+ bench3.1/video/langues/fr/32b.mp4 filter=lfs diff=lfs merge=lfs -text
466
+ bench3.1/video/langues/fr/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
467
+ bench3.1/video/langues/fr/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
468
+ bench3.1/video/langues/fr/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
469
+ bench3.1/video/langues/fr/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
470
+ bench3.1/video/langues/fr/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
471
+ bench3.1/video/langues/fr/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
472
+ bench3.1/video/langues/fr/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
473
+ bench3.1/video/langues/fr/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
474
+ bench3.1/video/langues/it/32b.mp4 filter=lfs diff=lfs merge=lfs -text
475
+ bench3.1/video/langues/it/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
476
+ bench3.1/video/langues/it/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
477
+ bench3.1/video/langues/it/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
478
+ bench3.1/video/langues/it/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
479
+ bench3.1/video/langues/it/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
480
+ bench3.1/video/langues/it/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
481
+ bench3.1/video/langues/it/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
482
+ bench3.1/video/langues/it/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
483
+ bench3.1/video/langues/ja/32b.mp4 filter=lfs diff=lfs merge=lfs -text
484
+ bench3.1/video/langues/ja/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
485
+ bench3.1/video/langues/ja/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
486
+ bench3.1/video/langues/ja/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
487
+ bench3.1/video/langues/ja/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
488
+ bench3.1/video/langues/ja/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
489
+ bench3.1/video/langues/ja/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
490
+ bench3.1/video/langues/ja/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
491
+ bench3.1/video/langues/ja/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
492
+ bench3.1/video/langues/ko/32b.mp4 filter=lfs diff=lfs merge=lfs -text
493
+ bench3.1/video/langues/ko/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
494
+ bench3.1/video/langues/ko/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
495
+ bench3.1/video/langues/ko/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
496
+ bench3.1/video/langues/ko/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
497
+ bench3.1/video/langues/ko/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
498
+ bench3.1/video/langues/ko/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
499
+ bench3.1/video/langues/ko/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
500
+ bench3.1/video/langues/ko/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
501
+ bench3.1/video/langues/pt/32b.mp4 filter=lfs diff=lfs merge=lfs -text
502
+ bench3.1/video/langues/pt/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
503
+ bench3.1/video/langues/pt/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
504
+ bench3.1/video/langues/pt/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
505
+ bench3.1/video/langues/pt/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
506
+ bench3.1/video/langues/pt/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
507
+ bench3.1/video/langues/pt/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
508
+ bench3.1/video/langues/pt/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
509
+ bench3.1/video/langues/pt/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
510
+ bench3.1/video/langues/ru/32b.mp4 filter=lfs diff=lfs merge=lfs -text
511
+ bench3.1/video/langues/ru/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
512
+ bench3.1/video/langues/ru/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
513
+ bench3.1/video/langues/ru/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
514
+ bench3.1/video/langues/ru/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
515
+ bench3.1/video/langues/ru/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
516
+ bench3.1/video/langues/ru/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
517
+ bench3.1/video/langues/ru/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
518
+ bench3.1/video/langues/ru/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
519
+ bench3.1/video/langues/zh/32b.mp4 filter=lfs diff=lfs merge=lfs -text
520
+ bench3.1/video/langues/zh/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
521
+ bench3.1/video/langues/zh/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
522
+ bench3.1/video/langues/zh/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
523
+ bench3.1/video/langues/zh/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
524
+ bench3.1/video/langues/zh/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
525
+ bench3.1/video/langues/zh/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
526
+ bench3.1/video/langues/zh/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
527
+ bench3.1/video/langues/zh/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -6,226 +6,214 @@ tags:
6
  - text-to-video
7
  - qwen3-vl
8
  - text-encoder
 
9
  base_model:
10
  - Comfy-Org/MiniMax-H3
11
  - Qwen/Qwen3-VL-4B-Instruct
 
12
  library_name: comfyui
13
  ---
14
 
15
  # ClipProj — MiniMax H3 conditioning from a Qwen3-VL-4B or 8B
16
 
17
  **Projection matrices that let a small Qwen3-VL replace the Qwen3-VL-32B text encoder of MiniMax H3.**
 
18
 
19
- **15.7 GB 4.9 GB of VRAM**, with no change to the diffusion model, the VAEs or the sampler. The 32B is nvfp4, the small encoders int8; the projection itself costs 52 MB to 604 MB depending on the file.
20
 
21
- <video controls src="https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/demo/chess-comparison.mp4"></video>
 
 
22
 
23
- Five renders of one scene, in this order: **4B matrix-only, 8B matrix-only, the 32B reference, 8B residual, 4B residual.** The reference sits in the middle, so each half is read against it.
 
 
 
 
24
 
25
- | conditioning | encoder | projection on card | total |
26
- |---|---|---|---|
27
- | Qwen3-VL-32B nvfp4 | 15.69 GB | — | **15.7 GB** |
28
- | Qwen3-VL-8B int8 + `v3-mlp` | 10.01 GB | 604 MB | **10.6 GB** |
29
- | Qwen3-VL-8B int8 + `v3` | 10.01 GB | 84 MB | **10.1 GB** |
30
- | Qwen3-VL-4B int8 + `v3-mlp` | 4.83 GB | 503 MB | **5.3 GB** |
31
- | Qwen3-VL-4B int8 + `v3` | 4.83 GB | 52 MB | **4.9 GB** |
32
 
33
- Two things this table makes explicit, because both would otherwise flatter the result. **The quantisations differ**: the 32B is nvfp4, the students are int8 — part of the size gap is format, not parameter count. And **the residual is not free**: the node loads the matrix in float32 and keeps the residual in the dtype it was saved in, so a `-mlp` file costs roughly half a gigabyte on the card where the plain matrix costs a rounding error.
34
-
35
- | | |
36
- |---|---|
37
- | diffusion model | `minimax_h3_fl2va_pruned_int8_convrot` |
38
- | turbo LoRA | `minimax_h3_fl2v_turbo_8step_v1.0_comfyui_bf16`, strength 1.0 |
39
- | sampler / scheduler | `res_multistep` / `simple` |
40
- | steps | 8 |
41
- | seed | 42 |
42
- | resolution | 16:9 at 0.8 MP, upscaled 2× by RTX Video Super Resolution ULTRA |
43
- | frames | 192 at 24 fps — 8.00 s |
44
- | video VAE | `minimax_h3_video_vae_int8_convrot` |
45
- | audio VAE | `minimax_h3_audio_vae_fp32` |
46
-
47
- **Only the projection changes between the five.** Everything else is identical, and on one machine the pipeline is deterministic — running the same configuration twice gives byte-identical decoded video and audio, verified by MD5 — so every difference you see comes from the projection and nothing else.
48
-
49
- ### You will not reproduce these files, and that is expected
50
-
51
- Run the demo prompt with seed 42 on your own machine and you will get the same scene, not the same file. **The result depends on the model of GPU the encoder runs on.**
52
-
53
- This came out of an unrelated test — checking that three loading modes gave the same output — and the cards happened to be at hand. Four of them is not a study, and none of this was the point of the exercise; it is written down because it would otherwise look like something is broken. Same prompt, same seed, same everything else:
54
-
55
- | card | decoded video MD5 |
56
- |---|---|
57
- | RTX 4070 | `1daf9be3…` |
58
- | RTX 3090 | `0a415964…` |
59
- | RTX 3060 | `dcb2f965…` |
60
- | RTX 4070 Ti SUPER | `b3b185e7…` |
61
-
62
- Four cards, four results. **Two different RTX 3090s gave byte-identical output**, so it is the model that decides, not the individual card — and not the architecture either, since the 3060 and the 3090 are both Ampere and disagree.
63
-
64
- The cause is small and the consequence is not. Encoding the same prompt on two cards gives conditioning that agrees to a **relative error of 7 × 10⁻⁷** — cosine 1.00000000, largest single-component difference 0.002. Different numbers of compute units mean different reduction orders, so floating-point additions do not happen in the same sequence. Eight denoising steps turn that into a different piece of furniture, or a wristwatch that is there on one card and absent on another. That watch is nowhere in the prompt, which is exactly why it is free to move.
65
-
66
- So: on one machine, with one card, everything here is reproducible to the bit — that is what makes the five-way comparison above meaningful. Across machines, expect the same scene with different details. This is a property of the diffusion model and its sampler, not of the projections: the reference 32B behaves identically.
67
-
68
- The full prompt is in [`demo/chess-prompt.txt`](demo/chess-prompt.txt), the settings above in machine-readable form in [`demo/generation-settings.json`](demo/generation-settings.json), and the five renders are in [`demo/`](demo) one by one if you want to step through them.
69
-
70
- **What to look at.** Everything the prompt states is there, on all five: the seated pose, the red dress, the white pieces on her side, the two captured black pawns, the cat, the straw hat, the laundry — and her knee, asked for three times and ending on a sentence of its own, *"Her knee never stops bouncing."* A continuous involuntary motion with no narrative purpose is the clearest single sign that a projection carried what was written, and it carries on the plain matrices too.
71
-
72
- **One thing none of the five gets right, the 32B included:** she lifts a knight and does not set it back on the same square, and on the 8B residual there is no knight on the board at all. Object permanence behind an occluding hand, on a grid of sixty-four identical squares, is a limit of the video model rather than of the conditioning.
73
-
74
- **And the terrace is furnished differently from one render to the next — that is not infidelity.** The prompt asks for a densely lived-in terrace without anchoring most of it: the cat is *"stretched out asleep in the sun"*, and nothing says where. What is left open, the model invents, and it invents differently depending on **the projection, the seed, and the model of GPU** — all three act on that same free space and none of them touches what was written. Two renders side by side give the impression of a different seed; that impression is what an unconstrained description looks like.
75
-
76
- > ⚠️ **Proof of concept — working, but a proof of concept.** It runs and produces good video, and every number below was measured on real hardware. Built and tested on a single setup (Windows 11, NVIDIA, ComfyUI 0.31.0) with deliberately limited exploration.
77
 
78
- > **Arriving from a tutorial or an article?** Anything published before this release names older files. Nothing has been deleted — the previous sets are still here — but take the current one: **`mmh3-4b-ClipProj-v3-mlp`** for a Qwen3-VL-4B, **`mmh3-8b-ClipProj-v3-mlp`** for an 8B. They need node **0.1.13 or later**, and an earlier node raises `KeyError: 'W'` on them rather than falling back quietly.
79
 
80
- These files are useless on their own. They require the custom node:
81
- **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
82
 
83
- ## Where this came from
 
84
 
85
- I am not an ML researcher. I work in imaging, and programming is a tool and a hobby rather than my trade. This started as something to tinker with: I wanted to understand how a diffusion model actually uses its text encoder, and the only way I know how to understand something is to take it apart and see whether it still runs afterwards.
 
 
 
 
86
 
87
- So the question was never "how do I save VRAM". It was "is this even possible at all". I expected it to fail. A linear map between two models that were never trained together, fitted in a single pass with no gradients and no learning rate, has no business producing usable video.
 
88
 
89
- It did, and the first results were good enough that keeping them on my own disk seemed silly. That is the whole story, and it is why this is labelled a proof of concept rather than a tool: it was never designed as one.
 
90
 
91
- It is also why there are so many measurements on the model card. Before showing this to anyone I had to convince myself I was not fooling myself, and most of what I tried along the way turned out to be wrong. Those attempts are written down as well, in [MEASUREMENTS.md](https://github.com/nicolab28/ComfyUI-ClipProj/blob/main/MEASUREMENTS.md) and [CALIBRATION.md](https://github.com/nicolab28/ComfyUI-ClipProj/blob/main/CALIBRATION.md).
92
 
93
- ## What changed in v3
94
 
95
- **Calibrated against the stock encoder.** The previous matrices were fitted against a modified 32B. While testing them we found that naming one part of a body could rewrite the whole of it — build, height and face shifting together, none of it asked for. v3 targets `qwen3vl_32b_minimax_h3_nvfp4_awq`, the encoder a plain `Load CLIP` gives you, and we no longer observe the problem.
96
 
97
- **The 8B now sees image tokens.** Until v3 the image corpus had only ever been encoded with a 4B student, so both 8B matrices projected vision tokens without having seen a single one — while the node accepts a reference image. Measured on 100 held-out images, on the raw conditioning the diffusion model actually receives:
 
 
 
 
 
98
 
99
- | | vision tokens | text in the same sequences |
100
- |---|---|---|
101
- | 8B residual, before | 0.7692 | 0.9085 |
102
- | 8B residual, after | 0.8578 | 0.9605 |
103
- | 8B matrix, before | 0.7845 | 0.8926 |
104
- | 8B matrix, after | 0.8457 | 0.9361 |
105
 
106
- That costs 0.0027 of pure-text cosine on the residual and 0.0013 on the matrix. Both students now share the same corpus, so the 4B and the 8B are comparable to each other for the first time.
 
107
 
108
- **Node 0.1.13 is required.** The `-v3-mlp` files carry no linear matrix at all the non-linear part does the whole job — and an earlier node raises `KeyError: 'W'` when it opens one.
 
 
 
109
 
110
- **And `ClipProjApply` now works with an int8 encoder.** Loading a quantised Qwen3-VL through ComfyUI's own `Load CLIP` and handing it to `ClipProjApply` used to fail inside the vision tower as soon as a reference image was present: `dequantize_int8_embedding` was called on a tensor the cast context had already dequantised to bf16, and the error named nothing useful. The position embedding now falls back to a plain lookup when that happens. A text-only prompt never triggered it, which is why it stayed hidden — it only appeared with an image, on the pageable path. The five demo renders above use that path.
 
111
 
112
- **One number, measured the same way for everything.** The `cos_test` written inside each file is its training-time figure against its own campaign's target, and it does **not** compare across versions. The comparable number is stored separately as `cos_prompt_reference`: every projection encoded against the same stock 32B, on the same prompt.
113
 
114
- | projection | vs 32B |
115
- |---|---|
116
- | `mmh3-8b-ClipProj-v3-mlp` | 0.9449 |
117
- | v2 `mmh3-8b-ClipProj-celeb-mlp` | 0.9393 |
118
- | `mmh3-4b-ClipProj-v3-mlp` | 0.9381 |
119
- | v2 `mmh3-4b-ClipProj-celeb-mlp` | 0.9293 |
120
- | `mmh3-8b-ClipProj-v3` | 0.9289 |
121
- | `mmh3-4b-ClipProj-v3` | 0.9193 |
122
 
123
- **Prefer the `-mlp` files, on the measurement — not on this scene.** They sit closer to the 32B, 0.9449 against 0.9289 on the 8B. But watch the video before assuming that shows: on this prompt the five renders are faithful, plain matrices included. The pose, the red dress, the white pieces, the cat, the straw hat, the laundry, the bouncing knee — all of it holds on all five.
 
 
124
 
125
- The only thing nobody gets right is the knight. The prompt asks her to lift one of her own pieces and set it back down without committing to the move; on every render it lands somewhere else, and on one of them — the 8B residual, the best-measuring file of the set — there is no knight on the board at all. That is object permanence behind an occluding hand on a grid of sixty-four identical squares, a limit of the video model rather than of the conditioning, and the 32B reference fails it too.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
126
 
127
- So the honest reading is: a tightly written prompt survives even the linear baseline, and the difference between the files shows up in the numbers well before it shows up on screen.
128
 
129
- ## What this is
130
 
131
- MiniMax H3 conditions on a Qwen3-VL-32B truncated to 50 layers 15.7 GB in NVFP4 — solely to turn a prompt into a `[seq, 5120]` tensor. This repository provides a learned map that lets a much smaller Qwen3-VL produce the same conditioning:
 
 
132
 
133
  ```
134
- cond = ((h - mean_in) / std_in) @ W * std_out + mean_out
 
 
 
 
 
 
 
135
  ```
136
 
137
- and, in the `-mlp` files, plus the output of a small residual network fed the same standardised input.
138
 
139
- It works because every Qwen3-VL shares the **same tokenizer** (151936 tokens): a prompt yields the same tokens at the same positions in both models, so a position-by-position mapping between their hidden states can be learned. The matrix is fitted by plain **ridge regression** — no gradients, no epochs, no learning rate. The residual network is the only part that is trained.
 
140
 
141
- ## Files
142
-
143
- Put them in `ComfyUI/models/clip_projections/`.
144
-
145
- **Start with `mmh3-8b-ClipProj-v3-mlp` if you have the VRAM, `mmh3-4b-ClipProj-v3-mlp` otherwise.** Both need node 0.1.13 or later.
146
-
147
- | File | Encoder | Structure | Size | vs 32B |
148
- |---|---|---|---|---|
149
- | `mmh3-4b-ClipProj-v3-mlp` | any Qwen3-VL-4B | non-linear | 503 MB | 0.9381 |
150
- | `mmh3-8b-ClipProj-v3-mlp` | any Qwen3-VL-8B | non-linear | 604 MB | **0.9449** |
151
- | `mmh3-4b-ClipProj-v3` | any Qwen3-VL-4B | matrix only | 26 MB | 0.9193 |
152
- | `mmh3-8b-ClipProj-v3` | any Qwen3-VL-8B | matrix only | 42 MB | 0.9289 |
153
- | `mmh3-ClipProj-control-zero` | — | control, run it once | 52 MB | — |
154
- | `mmh3-ClipProj-control-identity` | — | control, run it once | 52 MB | — |
155
-
156
- The v2 files — `mmh3-4b-ClipProj-celeb-mlp` and the seven beside it — are kept and still work. They were calibrated against a modified 32B and against a corpus containing no image tokens, so prefer v3.
157
-
158
- The last column is the one number measured identically for every row: same stock 32B, same prompt, cosine averaged token by token. It is also written inside each file as `cos_prompt_reference`.
159
-
160
- The `-v3-mlp` files are larger than the v2 residuals — 503 and 604 MB against 304 and 386 — because the hidden width went from 16 384 to 32 768. That is the whole reason for the extra download.
161
-
162
- Every matrix works on **any variant of its own size**: the measured cosine gap between a bf16-calibrated matrix applied to an abliterated fp8 encoder is 0.0023. You do not need the exact checkpoint a matrix was calibrated on. The 8B matrices need an 8B encoder though — 4096 input dimensions instead of 2560 — and the node checks the width and refuses a mismatch.
163
-
164
- ## Named people
165
-
166
- **This is what changed in 0.1.3, and it was a corpus problem.**
167
-
168
- The calibration corpus named a person on about 70 lines out of 8632, roughly 0.02 % of the training tokens. The directions of the hidden space that carry an identity were therefore constrained by nothing at all, and the fit put whatever minimised the error on landscape descriptions there. Named people came out as somebody else.
169
-
170
- The `-celeb` matrices add 500 people, ranked by popularity, with five short prompts and two long ones each. What it buys and what it costs:
171
-
172
- | | name tokens | rest of the sentence | general test set |
173
- |---|---|---|---|
174
- | without | 0.8265 | 0.9358 | 0.7944 |
175
- | with | **0.8844** | **0.9516** | 0.7930 |
176
-
177
- Seven thousandths of cosine on the general corpus, for six points on the tokens that carry an identity. The rest of the sentence improves too, because the celebrity prompts are short and the general corpus had nothing under fifteen words.
178
-
179
- Two findings that decide how far this is worth pushing.
180
-
181
- **Two contexts per person are enough.** Measured on contexts held out for people the matrix had seen: 0.9875 at two, 0.9945 at five, 0.9986 at twenty. Forty is a waste.
182
-
183
- **Five hundred names generalise to names never seen.** A held-out band at popularity ranks 501 to 540, absent from every calibration, reconstructs at 0.8795 against 0.8844 for the covered ones. Covering 500 people does not teach 500 names; it teaches the map how to handle that region of the space. Going to several thousand would buy very little.
184
 
185
- **What still fails is not the corpus.** Characters whose identity is a mask rather than a face come out as a stranger wearing the right costume. People whose fame predates the era when everything was photographed come out wrong or generic. And some names fail on the plain 32B too, so run the reference before blaming the projection — that check has overturned three of my own conclusions.
 
186
 
187
- ## Where the calibration data comes from
188
 
189
- The general corpus is [GokuScraper/seedance-2-prompts-datasets](https://huggingface.co/datasets/GokuScraper/seedance-2-prompts-datasets), filtered to prompts of fifteen words or more and deduplicated: 8632 lines, median 128 words. The 500 named people come from a TMDB export published on Kaggle, ranked by popularity, with transliterated names dropped beyond rank 1000.
 
 
 
 
 
 
 
190
 
191
- Around each name, five short prompts are generated from templates, and two longer ones in MiniMax H3's section format are written by Mistral Small and Gemini Flash Lite, half each. Everything needed to rebuild the corpus is in the node's `calibration/` folder, including the system prompt the long prompts were written from.
192
 
193
- ## The residual network
 
194
 
195
- The `-mlp` files carry a `d_in 16384 → 5120` network with a GELU, added to the matrix rather than replacing it. Its last layer is initialised to zero, so at the first step the model reproduces the matrix exactly and can only improve on it. It is worth 0.05 to 0.08 of cosine, four times what multiplying the corpus by eleven buys the linear map.
 
 
 
 
 
 
 
196
 
197
- **Which of the two renders better is not settled.** The cosine does not predict it that is the single most repeated lesson of this project. Try both on your own prompts.
 
 
198
 
199
- Two things measured while building it. Width beats depth: at equal parameter count, two hidden layers of 8192 reach 0.7691 against 0.7944 for one layer of 16384. And a residual extrapolates worse than a matrix does — outside the corpus it saw, a linear map degrades gracefully while the network collapses.
 
 
 
200
 
201
- ## Measured results
202
 
203
- > **Two scales of cosine appear on this page and they do not compare.** The table just below is the v2 training-time figure, measured on held-out prompts against *that* campaign's target. The 0.93–0.94 figures higher up are `cos_prompt_reference`: one prompt, every projection encoded against the same stock 32B. A number is only ever comparable to another measured the same way — mixing them is how a set of matrices can look like it improved when nothing was established.
 
 
 
204
 
205
- The v2 campaign, on its own held-out set:
206
 
207
- | | 4B | 8B |
208
- |---|---|---|
209
- | matrix, no names | 0.7169 | 0.7528 |
210
- | matrix + residual | 0.7944 | 0.7970 |
211
- | matrix, names covered | 0.7095 | 0.7466 |
212
- | matrix + residual, names covered | 0.7930 | **0.8037** |
213
 
214
- A cosine of 0.79 sounds poor and is not — the DiT tolerates far more than the metric suggests. What holds up in actual generation: simple prompts, structured multi-shot prompts with several distinct cuts and no bleed between them, fl2va with first and last frame, ref2va with a reference image, and since 0.1.3 ref2va with a reference video.
215
 
216
- Fidelity does **not** collapse on short prompts: measured per-token cosine goes from 0.937 at 80 words to 0.908 at 2 words, once the attention sink is handled.
217
 
218
- ## Speech
 
 
219
 
220
- The first release lost non-English speech: a French line came out half Spanish, and the 8B put everything in English. That was the clearest regression and I could not explain it then.
 
 
221
 
222
- With `mmh3-8b-ClipProj-celeb-mlp`, a three-shot clip carrying English, French and Spanish comes out like the 32B does, and the audio level gap measured against the reference has gone from 7.6 dB to 3.5.
 
223
 
224
- Part of what was blamed on the projection was not the projection. A line that fills more than about two thirds of its shot comes out slurred whatever encoder produced the conditioning — the fix is a longer shot, not a better matrix. And MiniMax H3 expects speech wrapped in `<d>[Language] ...</d>` with a stable speaker id declared beforehand; without that, one voice with one accent is used for the whole clip. Neither of those is documented here because neither is ours, but both cost me a day.
 
 
 
225
 
226
  ## Run the controls first
227
 
228
- The two control matrices exist to prove the learned matrix is doing the work rather than the diffusion model. Same prompt, same seed, only the matrix changes:
 
229
 
230
  | Matrix | Output for *"a red ball on a wood table"* |
231
  |---|---|
@@ -233,60 +221,53 @@ The two control matrices exist to prove the learned matrix is doing the work rat
233
  | `mmh3-ClipProj-control-identity` | a golden object in flames — unusable |
234
  | a learned matrix | the red ball on a wood table |
235
 
236
- `‖W_identity‖ = 50.6` against `‖W_learned‖ = 52.4` — near-identical energy, so the difference is structural, not a matter of scale.
237
-
238
- **If the identity control ever looks fine, the learned matrix adds nothing — and you want to know that before trusting it.**
239
-
240
- ## What is in obsolete/
241
-
242
- The previous matrices, kept because a comparison posted on r/StableDiffusion ran on them and the links have to keep working. They have no name coverage and are calibrated on a corpus thirty times smaller. There is no reason to prefer them.
243
 
244
- Among them, the `CONDPROJ` pair, and the story is worth telling because the mistake was instructive.
245
 
246
- The DiT does not consume the conditioning as it arrives: it first passes it through `condition_proj`, a `Linear(5120 → 5376)` feeding the token refiner. That layer's spectrum is very uneven — a factor of 45 between the top and bottom deciles of its singular values, 52 % of the energy in 10 % of the directions. Plain ridge regression ignores this and spends as much effort on a direction the DiT will multiply by 0.10 as on one it will multiply by 37. Calibrating against the **output** of that layer instead, then mapping back through the pseudo-inverse, should therefore minimise the error the DiT actually sees. The cosine went from 0.697 to 0.845 on the 4B and 0.731 to 0.860 on the 8B.
 
247
 
248
- Then I compared what the two matrices actually output:
 
249
 
250
- ```
251
- 4B CONDPROJ against unweighted, same corpus cosine 0.999998
252
- 8B CONDPROJ against unweighted, same corpus cosine 0.999999
253
- ```
254
-
255
- They are the same function. Unregularised least squares is invariant to an invertible linear transform of the targets, so fitting in one space and mapping back recovers the same map; only the ridge penalty breaks that invariance, and with 37 851 training tokens against λ = 1000 it barely binds. The entire gain was an artefact of measuring in a different space.
256
-
257
- *The idea came from u/stddealer on r/StableDiffusion, and it was a good one. The measurement is on me: I published the cosine before checking whether the matrix had changed at all.*
258
 
259
- ## Known limitations
 
260
 
261
- **Quantisation costs facts.** Comparing `int8_convrot` against `bf16` on factual recall shows errors appearing under quantisation. Fine for general use, worth knowing if your prompts lean on proper nouns.
262
 
263
- **Masks defeat identity.** A character recognised by a costume rather than a face comes out as an unknown person in the right suit. No corpus fixes that, because the identity is not in the name's representation to begin with.
264
-
265
- **Counting is unreliable, and not because of the projection.** Ask for three of something and you get four, on the 32B too. Enumerating works better than announcing a number.
266
 
267
  ## Required models
268
 
269
  | Role | Model |
270
  |---|---|
271
  | Diffusion model + VAEs | [Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3) |
272
- | Text encoder, 4B | [Comfy-Org/Krea-2](https://huggingface.co/Comfy-Org/Krea-2) → `text_encoders/qwen3vl_4b_fp8_scaled.safetensors` |
273
- | Text encoder, 8B | any ComfyUI-format Qwen3-VL-8B (the 8B matrices expect 4096 input dims) |
274
 
275
  The 32B text encoder is **no longer needed** — that is the entire point.
276
 
277
  ## Licence and responsibility
278
 
279
- These matrices are released under **MIT**, like the node.
280
-
281
- They are derived from the activations of both models, and their legal status is unclear. They are provided as-is, for research, with no claim of ownership over anything derived from the underlying models.
282
-
283
- - **Qwen3-VL** is published by Alibaba under **Apache 2.0**. Read and comply with its terms and acceptable-use policy.
284
- - **MiniMax H3** ships under a **custom licence**. Read it before any use, particularly commercial.
285
 
286
- This project is **not affiliated with, endorsed by, or connected to** Alibaba / Qwen, MiniMax, or Comfy Org.
 
287
 
288
- You remain responsible for what you generate and for complying with the licences of every model you load.
 
289
 
290
  ## Credits
291
 
292
- Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number quoted was measured on real hardware, not estimated: where a prediction turned out wrong, the measurement won and the text was corrected. Three claims in the previous version of this file were wrong and are corrected here.
 
 
 
 
6
  - text-to-video
7
  - qwen3-vl
8
  - text-encoder
9
+ - multilingual
10
  base_model:
11
  - Comfy-Org/MiniMax-H3
12
  - Qwen/Qwen3-VL-4B-Instruct
13
+ - Qwen/Qwen3-VL-8B-Instruct
14
  library_name: comfyui
15
  ---
16
 
17
  # ClipProj — MiniMax H3 conditioning from a Qwen3-VL-4B or 8B
18
 
19
  **Projection matrices that let a small Qwen3-VL replace the Qwen3-VL-32B text encoder of MiniMax H3.**
20
+ **15.0 GB → 4.6 GB**, with no change to the diffusion model, the VAEs or the sampler.
21
 
22
+ <video controls width="360" src="https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4"></video>
23
 
24
+ *Eleven languages, 88 seconds. For each one, the **smallest** file that matches the 32B — not the best
25
+ one. Nine of the eleven run on a 4B.
26
+ [Direct link](https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4)*
27
 
28
+ > ⚠️ **The video looks rough, and that is on purpose.** It is rendered at 0.3 MP with 6 sampling steps
29
+ > the settings that made 297 renders affordable — then upscaled. The audio is tinny for the same reason,
30
+ > **identically so on the 32B**, with the same 19 dB dip between 1 and 3 kHz. This is a pronunciation test,
31
+ > not a showcase: it exists to let you hear *which words come out*, not how pretty the result is. Render at
32
+ > your own settings and it will look like MiniMax H3 normally looks.
33
 
34
+ Requires the custom node **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**,
35
+ version 0.1.13 or later. The v3.1 files need no code change — same base as v3.
 
 
 
 
 
36
 
37
+ ---
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38
 
39
+ ## Read this before the tables: these are metrics, not verdicts
40
 
41
+ **I do not speak these eleven languages.** I cannot tell you whether a render sounds right, and I have not
42
+ asked anyone who can. No native speaker has listened to any of the 297 speech renders behind this page.
43
 
44
+ Every figure below is a **distance between two automatic transcriptions** — what one machine wrote down
45
+ from the reference, against what it wrote down from the projection. That is all it is.
46
 
47
+ **What that captures.** Whether the same words and the same sounds come out. Two instruments are used
48
+ because each is wrong in a known direction: **Whisper** has a language model inside and corrects a slurred
49
+ word into the most probable real one, so it *under*-reports defects — a lower bound. **ZIPA** has no
50
+ lexical decoder and counts every shift in realisation as an error, so it *over*-reports — an upper bound.
51
+ What a listener would notice lies between them.
52
 
53
+ **What it does not capture.** Prosody, rhythm, timbre, naturalness. A file scoring 98.8 here could still
54
+ sound foreign to someone who speaks the language.
55
 
56
+ **So read a score as "close to the 32B, according to this instrument" never as "good".** If you speak one
57
+ of these languages, your ear outranks every table here, and I would genuinely like to hear what it says.
58
 
59
+ ---
60
 
61
+ ## Files take a v3.1
62
 
63
+ Put them in `ComfyUI/models/clip_projections/`.
64
 
65
+ | File | Encoder | Head | Projection | Encoder + projection |
66
+ |---|---|---|---|---|
67
+ | **`mmh3-4b-ClipProj-v3.1`** | any Qwen3-VL-4B | ridge | 26 MB | **4.6 GB** |
68
+ | `mmh3-4b-ClipProj-v3.1-mlp` | any Qwen3-VL-4B | residual | 481 MB | 5.1 GB |
69
+ | `mmh3-8b-ClipProj-v3.1` | any Qwen3-VL-8B | ridge | 41 MB | 9.6 GB |
70
+ | `mmh3-8b-ClipProj-v3.1-mlp` | any Qwen3-VL-8B | residual | 577 MB | 10.1 GB |
71
 
72
+ **Start with `mmh3-4b-ClipProj-v3.1`.** Nine of the eleven benchmarked languages run on it, and nothing
73
+ distinguishes it from the 32B in image generation. The 8B earns its extra 5 GB on Arabic, French, and
74
+ proper nouns generally.
 
 
 
75
 
76
+ The 8B matrices expect 4096 input dimensions instead of 2560; the node checks the width and refuses a
77
+ mismatch. Every matrix works on any variant of its own size.
78
 
79
+ **One format detail.** The two `-mlp` files carry **no `W` tensor**they were trained without a linear
80
+ path, so the residual network carries everything, and a matrix of zeros would cost 26 MB plus a 2560×5120
81
+ matmul per token to add nothing. The node treats `W` as optional and reports `| residual only`. This is
82
+ expected, not a truncated download.
83
 
84
+ The earlier `-celeb` and v3 files remain available. There is no reason to prefer them: the benchmark
85
+ separates v3 from v3.1 cleanly, and in the same direction on every metric.
86
 
87
+ ---
88
 
89
+ ## What changed in v3.1: giving every script its share
 
 
 
 
 
 
 
90
 
91
+ v3 was calibrated on a corpus that was overwhelmingly English, with the other languages bolted on
92
+ afterwards as a top-up. v3.1 **adds text and tagged prompts until every writing system carries roughly
93
+ comparable weight** — English excepted, since the prompt format itself is English.
94
 
95
+ | Script | Languages | Tagged prompts | Raw text | Share of corpus |
96
+ |---|---|---|---|---|
97
+ | Latin — base | English: original corpus, image and register lots | *base* | — | **68.3 %** |
98
+ | Han | zh | ✓ | ✓ | 6.7 % |
99
+ | Hangul | ko | ✓ | ✓ | 6.6 % |
100
+ | Latin, accented | fr | ✓ | ✓ | 6.3 % |
101
+ | Arabic | ar | ✓ | ✓ **(new)** | 4.1 % |
102
+ | Latin | es, de, it, pt | ✓ | — | 1.3 % each |
103
+ | Cyrillic | ru | ✓ | — | 1.3 % |
104
+
105
+ The rule behind those numbers: **a script that inherits nothing from Latin needs raw text**; a Latin script
106
+ only needs tagged prompts, because the alphabet is already covered. The one genuinely new lot is raw
107
+ Arabic — 550 000 characters, the last non-Latin script still living on tagged prompts alone.
108
+
109
+ Training also restarted from scratch rather than topping up: a network keeps the order it learned in, and
110
+ lowering the learning rate on a top-up only arbitrates between preserving and correcting. Architecture and
111
+ hyper-parameters are identical to v3 — `hidden 32768`, `depth 1`, `tap 24`, `lr 1e-3` — so what the
112
+ benchmark compares is the corpus, not the recipe.
113
 
114
+ ---
115
 
116
+ ## The benchmark
117
 
118
+ Everything below comes from **three seeds**42, 100 000 and 100 000 000 because a single draw cannot
119
+ separate a real gap from chance. All of it is in [`bench3.1/`](./bench3.1): 297 speech renders, 405 image
120
+ renders, the raw CSVs and the montage.
121
 
122
  ```
123
+ bench3.1/
124
+ README.md this measurement report in full
125
+ audio/<lang>/ 297 FLAC — 9 conditionings × 11 languages × 3 seeds
126
+ video/langues/<lang>/ 99 renders, one per conditioning
127
+ video/ the eleven-language montage
128
+ images/brutes/ 75 PNG — 15 scenes × 5 conditionings
129
+ images/planches/ 15 comparison sheets, five renders side by side
130
+ mesures/ phonemes, words, image cosines, scores, prompts
131
  ```
132
 
133
+ ### The reference is not perfection
134
 
135
+ A cosine of 0.79 or "23 character errors" has no scale, and zero errors is not the target either, because
136
+ **the 32B does not reproduce itself**. Change nothing but the seed:
137
 
138
+ | | 32B against itself |
139
+ |---|---|
140
+ | Speech | **5.8 phonemes out of 75** (7.8 %) |
141
+ | Image | **0.9552** SigLIP2 cosine (floor 0.5313) |
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
142
 
143
+ That gap is the unit. Everything is normalised so **32B = 100**: at 100, swapping the encoder moves the
144
+ output as much as changing the seed does.
145
 
146
+ ### Results
147
 
148
+ | Conditioning | Speech | ± | Image | Prompt | Marseille |
149
+ |---|---|---|---|---|---|
150
+ | **32B** *(reference)* | **100.0** | — | **100.0** | **100.0** | 3/3 |
151
+ | `8b-ClipProj-v3.1` | **98.8** | ±1.4 | 100.0 | 100.4 | 3/3 |
152
+ | `8b-ClipProj-v3.1-mlp` | 98.2 | ±1.2 | 101.4 | **102.3** | 3/3 |
153
+ | `4b-ClipProj-v3.1` | 97.9 | ±1.0 | 100.1 | 99.4 | 0/3 |
154
+ | `4b-ClipProj-v3.1-mlp` | 97.8 | ±1.1 | 100.9 | 100.3 | 1/3 |
155
+ | *the four v3 files* | *93.2 – 95.6* | | *99.4 – 102.4* | *99.0 – 100.7* | *0–3/3* |
156
 
157
+ `±` is the spread across the three seeds. **Two files separated by less than that are not separated.**
158
 
159
+ The raw counts, three units never added together — PER over three seeds, WER and CER as Whisper hears
160
+ them, single-seed:
161
 
162
+ | | PER | | WER | CER |
163
+ |---|---|---|---|---|
164
+ | *32B against itself* | *≈193 / 2469* | *7.8 %* | — | — |
165
+ | `8b-v3.1` | 211 / 2469 | 8.5 % | 10 / 174 | 14 / 869 |
166
+ | `8b-v3.1-mlp` | 222 / 2469 | 9.0 % | 15 / 174 | 28 / 869 |
167
+ | `4b-v3.1-mlp` | 230 / 2469 | 9.3 % | 13 / 174 | 23 / 869 |
168
+ | `4b-v3.1` | 232 / 2469 | 9.4 % | 15 / 174 | 30 / 869 |
169
+ | the four v3 | 281–341 / 2469 | 11.4–13.8 % | 17–23 / 174 | 29–46 / 869 |
170
 
171
+ **What separates: the corpus.** The four v3.1 files land within one point of each other while ranging from
172
+ 4.6 to 10.1 GB. The four v3 sit a clear notch below at identical sizes. A 4B v3.1 beats an 8B v3 by four
173
+ points while weighing half as much.
174
 
175
+ **What does not separate: everything else.** Not 4B against 8B on general pronunciation, not ridge against
176
+ residual, and nothing at all in image generation — all nine conditionings, v3 included, sit between 99.4
177
+ and 102.4 there, where the standard deviation within a single model is three to five times the entire
178
+ spread between models.
179
 
180
+ ### The one place 4B and 8B part company
181
 
182
+ The French line says *"la lumière de **Marseille**"*. Six phonemes out of seventy the error rate drowns
183
+ them, the ear does not. Across three seeds, the 32B and three 8B files say it every time; the best 4B says
184
+ it once out of three. Treat that for what it is: **one proper noun, in one language out of eleven**. It is
185
+ not a ranking criterion, but if your prompts lean on names, test both.
186
 
187
+ ### Full report
188
 
189
+ [`bench3.1/README.md`](./bench3.1/README.md) carries the whole thing: per-language tables, the raw-text
190
+ correlation that does *not* hold, why the phoneme metric misleads in Portuguese and Russian, the
191
+ self-consistency measurements, and every limitation I know of.
 
 
 
192
 
193
+ ---
194
 
195
+ ## What this is, technically
196
 
197
+ MiniMax H3 conditions on a Qwen3-VL-32B truncated to 50 layers — 15.0 GB in NVFP4 — solely to turn a
198
+ prompt into a `[seq, 5120]` tensor. This repository provides a learned map so a much smaller Qwen3-VL can
199
+ produce the same conditioning:
200
 
201
+ ```
202
+ cond = ((h - mean_in) / std_in) @ W * std_out + mean_out
203
+ ```
204
 
205
+ plus, in the `-mlp` files, the output of a residual network fed the same standardised input and in the
206
+ v3.1 residuals, *only* that network.
207
 
208
+ It works because every Qwen3-VL shares the **same tokenizer** (151936 tokens): a prompt yields the same
209
+ tokens at the same positions in both models, so a position-by-position mapping between hidden states can
210
+ be learned. The matrix is fitted by plain **ridge regression** — no gradients, no epochs. Only the residual
211
+ network is trained.
212
 
213
  ## Run the controls first
214
 
215
+ Two control matrices prove the learned matrix is doing the work rather than the diffusion model. Same
216
+ prompt, same seed, only the matrix changes:
217
 
218
  | Matrix | Output for *"a red ball on a wood table"* |
219
  |---|---|
 
221
  | `mmh3-ClipProj-control-identity` | a golden object in flames — unusable |
222
  | a learned matrix | the red ball on a wood table |
223
 
224
+ `‖W_identity‖ = 50.6` against `‖W_learned‖ = 52.4` — near-identical energy, so the difference is
225
+ structural, not a matter of scale. **If the identity control ever looks fine, the learned matrix adds
226
+ nothing, and you want to know that before trusting it.**
 
 
 
 
227
 
228
+ ## Limitations
229
 
230
+ **Three seeds fix the order of magnitude of the noise, not its tail.** Any gap under one point of score is
231
+ not a result.
232
 
233
+ **The cosine is blind to countable attributes.** A whole loaf and a halved loaf, same crust, same paper,
234
+ same light, give the same vector to the fourth decimal. Image equivalence here means *global appearance*.
235
 
236
+ **The phoneme metric is unreliable in Portuguese, Russian and Korean** — not the speech itself. The
237
+ reference drifts by 17.0, 10.7 and 11.3 phonemes there between seeds, while Whisper transcribes the same
238
+ renders with zero to three character errors.
 
 
 
 
 
239
 
240
+ **Speech quality is deliberately poor**, as the video shows — 6 to 8 steps, identically for the 32B. The
241
+ benchmark measures correctness of pronunciation, not fidelity of reproduction.
242
 
243
+ **Quantisation costs facts**, which is the most likely explanation for the proper-noun gap.
244
 
245
+ **Masks defeat identity**, and **counting is unreliable** both true on the 32B too.
 
 
246
 
247
  ## Required models
248
 
249
  | Role | Model |
250
  |---|---|
251
  | Diffusion model + VAEs | [Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3) |
252
+ | Text encoder, 4B | any ComfyUI-format Qwen3-VL-4B |
253
+ | Text encoder, 8B | any ComfyUI-format Qwen3-VL-8B |
254
 
255
  The 32B text encoder is **no longer needed** — that is the entire point.
256
 
257
  ## Licence and responsibility
258
 
259
+ MIT, like the node. These matrices are derived from the activations of both models and their legal status
260
+ is unclear; they are provided as-is, for research.
 
 
 
 
261
 
262
+ - **Qwen3-VL** Alibaba, Apache 2.0. Read its terms and acceptable-use policy.
263
+ - **MiniMax H3** — custom licence. Read it before any use, particularly commercial.
264
 
265
+ Not affiliated with, endorsed by, or connected to Alibaba / Qwen, MiniMax, or Comfy Org. You remain
266
+ responsible for what you generate and for complying with the licences of every model you load.
267
 
268
  ## Credits
269
 
270
+ Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number quoted was measured on this hardware,
271
+ never estimated. Where a prediction lost to a measurement, the measurement won and the text was rewritten
272
+ — which happened several times in this release, the largest being a single-seed ranking of the v3.1 files
273
+ that dissolved entirely once the 32B's own variance was known.
bench3.1/README.md ADDED
@@ -0,0 +1,494 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ tags:
4
+ - comfyui
5
+ - minimax-h3
6
+ - text-to-video
7
+ - qwen3-vl
8
+ - text-encoder
9
+ - multilingual
10
+ base_model:
11
+ - Comfy-Org/MiniMax-H3
12
+ - Qwen/Qwen3-VL-4B-Instruct
13
+ - Qwen/Qwen3-VL-8B-Instruct
14
+ library_name: comfyui
15
+ ---
16
+
17
+ # ClipProj v3.1 — measured against the 32B's own variance
18
+
19
+ **Four projection matrices that let a Qwen3-VL-4B or 8B replace the Qwen3-VL-32B text encoder of MiniMax H3.**
20
+
21
+ **15.0 GB → 4.6 GB**, with no change to the diffusion model, the VAEs or the sampler.
22
+
23
+ This release is not about a new architecture. It is about finally knowing **how good these things are**, because the previous numbers could not tell me. This card is mostly the measurement, and the measurement changed three of my own conclusions.
24
+
25
+ Requires the custom node: **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
26
+
27
+ ---
28
+
29
+ ## Read this before the tables: these are metrics, not verdicts
30
+
31
+ **I do not speak these eleven languages.** I cannot tell you whether a render sounds right, and I have not
32
+ asked anyone who can. No native speaker has listened to any of the 297 speech renders on this page.
33
+
34
+ So nothing below is a judgement of quality. Every figure is a **distance between two automatic
35
+ transcriptions** — what one machine wrote down from the reference, against what it wrote down from the
36
+ projection. That is all it is, and it is worth being explicit about what that does and does not capture:
37
+
38
+ **What the numbers do capture.** Whether the same words and the same sounds come out. Two instruments are
39
+ used precisely because each is wrong in a known direction: **Whisper** has a language model inside and
40
+ corrects a slurred word into the most probable real one, so it *under*-reports pronunciation defects — a
41
+ lower bound. **ZIPA** has no lexical decoder at all and counts every shift in realisation as an error, so
42
+ it *over*-reports — an upper bound. What a listener would notice lies between them, and neither number
43
+ alone is the answer.
44
+
45
+ **What they do not capture.** Prosody, rhythm, timbre, naturalness — everything that makes speech sound
46
+ native rather than merely correct. A render scoring 98.8 here could still sound foreign to someone who
47
+ speaks the language. These metrics cannot see that, and neither can I.
48
+
49
+ **So read a score as "close to the 32B, according to this instrument"** — never as "good". If you speak
50
+ one of these languages, your ear outranks every table below, and I would genuinely like to hear what it
51
+ tells you.
52
+
53
+ ---
54
+
55
+ ## What changed in v3.1: giving every script its share
56
+
57
+ v3 was calibrated on a corpus that was overwhelmingly English, with the other languages bolted on
58
+ afterwards as a top-up. v3.1 **adds text and tagged prompts until every writing system carries roughly
59
+ comparable weight** — English excepted, because the prompt format itself is English: the sections, the
60
+ tags and the descriptions are all written in it, so it stays the majority no matter what.
61
+
62
+ Measured share of the v3.1 corpus:
63
+
64
+ | Script | Languages | Tagged prompts | Raw text | Share |
65
+ |---|---|---|---|---|
66
+ | Latin — base | English: the original corpus, image lots and register lots | *base* | — | **68.3 %** |
67
+ | Han | zh | ✓ | ✓ | 6.7 % |
68
+ | Hangul | ko | ✓ | ✓ | 6.6 % |
69
+ | Latin, accented | fr | ✓ | ✓ | 6.3 % |
70
+ | Arabic | ar | ✓ | ✓ **(new)** | 4.1 % |
71
+ | Latin | es, de, it, pt | ✓ | — | 1.3 % each |
72
+ | Cyrillic | ru | ✓ | — | 1.3 % |
73
+
74
+ The rule behind those numbers: **a script that inherits nothing from Latin needs raw text**; a Latin
75
+ script only needs tagged prompts, because the alphabet is already covered and roughly 250 tags are
76
+ enough to attach a language to it. That is why Chinese, Korean and French carry raw lots and Spanish
77
+ does not.
78
+
79
+ **The one genuinely new lot is raw Arabic** — 550 000 characters. Arabic was the last non-Latin script
80
+ still living on tagged prompts alone.
81
+
82
+ The training itself was also restarted from scratch rather than topped up. A network keeps the order it
83
+ learned in: whatever comes last weighs more, and lowering the learning rate on a top-up run does not
84
+ remove that imbalance, it only arbitrates between preserving what was acquired and correcting it. Same
85
+ architecture and same hyper-parameters as v3 — `hidden 32768`, `depth 1`, `tap 24`, `lr 1e-3`, no linear
86
+ path — so what the benchmark below compares is the corpus, not the recipe.
87
+
88
+ ### What it buys
89
+
90
+ Phoneme errors against the 32B, averaged over the four files of each generation, the three seeds and
91
+ compared against the threshold:
92
+
93
+ | | threshold | v3 | **v3.1** | |
94
+ |---|---|---|---|---|
95
+ | es | 0.0 | 1.9 | **0.5** | −74 % |
96
+ | de | 2.7 | 9.8 | **3.6** | −64 % |
97
+ | fr | 4.0 | 8.2 | **3.0** | −64 % |
98
+ | it | 2.3 | 4.2 | **1.6** | −63 % |
99
+ | ru | 10.7 | 21.5 | **13.6** | −37 % |
100
+ | ar | 6.3 | 11.1 | **7.6** | −32 % |
101
+ | zh | 3.3 | 3.0 | **2.2** | −25 % |
102
+ | ja | 6.7 | 7.5 | **6.4** | −14 % |
103
+ | ko | 11.3 | 12.8 | **11.5** | −10 % |
104
+ | pt | 17.0 | 25.1 | **24.1** | −4 % |
105
+ | en | 0.0 | **0.2** | 0.5 | +0.3 |
106
+ | **mean** | 5.8 | **9.6** | **6.8** | **−29 %** |
107
+
108
+ The European languages, which v3 only ever saw as a top-up, gain 60 to 74 %. Arabic gains 32 %, which is
109
+ where the new raw-text lot shows up. Russian gains 37 %.
110
+
111
+ **Raw text does not predict the outcome.** Rapported to each language's own threshold, the two groups
112
+ overlap completely:
113
+
114
+ | | script | raw text | threshold | v3 | v3.1 | × threshold |
115
+ |---|---|---|---|---|---|---|
116
+ | zh | Han | ✓ | 3.3 | 3.0 | 2.2 | **0.67** |
117
+ | it | Latin | — | 2.3 | 4.2 | 1.6 | **0.68** |
118
+ | fr | Latin | ✓ | 4.0 | 8.2 | 3.0 | 0.75 |
119
+ | ja | Kana/Kanji | — | 6.7 | 7.5 | 6.4 | 0.96 |
120
+ | ko | Hangul | ✓ | 11.3 | 12.8 | 11.5 | 1.01 |
121
+ | ar | Arabic | ✓ | 6.3 | 11.1 | 7.6 | 1.20 |
122
+ | ru | Cyrillic | — | 10.7 | 21.5 | 13.6 | 1.27 |
123
+ | de | Latin | — | 2.7 | 9.8 | 3.6 | 1.34 |
124
+ | pt | Latin | — | 17.0 | 25.1 | 24.1 | 1.42 |
125
+
126
+ Languages with a raw lot run 0.67 to 1.20; languages without run 0.68 to 1.42. Italian, with tagged
127
+ prompts only, lands second best overall.
128
+
129
+ **Russian is the clearest case.** It is the only non-Latin script here with no raw text and nothing to
130
+ inherit — Japanese borrows kanji from the Chinese lots, Latin scripts borrow the alphabet from English —
131
+ and it still gains **37 %** between v3 and v3.1 on tagged prompts alone, finishing ahead of German and
132
+ Portuguese, which are Latin. The rule written in the build scripts — *250 tags are enough once the
133
+ alphabet is covered* — evidently extends to Cyrillic, which the Qwen3-VL tokenizer covers natively.
134
+
135
+ So raw text is what an **unseen script** needs, not what a language needs. Where a script is already in
136
+ the tokenizer's reach, tags carry it.
137
+
138
+ **English pays for it, and the bill is half a phoneme out of 77.** That is the whole cost of rebalancing:
139
+ v3 was 0.2 errors, v3.1 is 0.5, both far below anything audible and below what a single seed resolves.
140
+ Portuguese barely moves, but nothing moves in Portuguese — the reference itself scatters by 17 there.
141
+
142
+ The net effect is a change of category rather than a better score. v3 sits at **1.46 to 1.77 times** the
143
+ threshold; v3.1 sits at **1.09 to 1.20**. From measurably worse than a seed change, to indistinguishable
144
+ from one.
145
+
146
+ ---
147
+
148
+ ## The problem with every number I published before
149
+
150
+ A cosine of 0.79, or "23 character errors out of 869" — neither has a scale. Is 23 good? Compared to what? Zero errors is not the right target either, because **the 32B does not reproduce itself**. Change nothing but the seed and it re-pronounces the sentence differently.
151
+
152
+ So the reference is not perfection. It is the 32B compared to itself, same prompt, different seed:
153
+
154
+ | | 32B against itself |
155
+ |---|---|
156
+ | Speech | **5.8 phonemes out of 75** (7.8 %) |
157
+ | Image | **0.9552** SigLIP2 cosine (floor: 0.5313) |
158
+
159
+ That gap is the unit. Everything below is normalised so that **32B = 100**:
160
+
161
+ - **100** — swapping the encoder moves the output as much as changing the seed
162
+ - **above 100** — it moves it less
163
+ - **below 100** — it moves it more
164
+
165
+ Below that threshold you are no longer measuring the projection. You are measuring the generator.
166
+
167
+ ---
168
+
169
+ ## Files
170
+
171
+ Put them in `ComfyUI/models/clip_projections/`.
172
+
173
+ | File | Encoder | Head | Size | Encoder + projection |
174
+ |---|---|---|---|---|
175
+ | `mmh3-4b-ClipProj-v3.1` | any Qwen3-VL-4B | ridge | 26 MB | **4.6 GB** |
176
+ | `mmh3-4b-ClipProj-v3.1-mlp` | any Qwen3-VL-4B | ridge + residual | 481 MB | 5.1 GB |
177
+ | `mmh3-8b-ClipProj-v3.1` | any Qwen3-VL-8B | ridge | 41 MB | 9.6 GB |
178
+ | `mmh3-8b-ClipProj-v3.1-mlp` | any Qwen3-VL-8B | ridge + residual | 577 MB | 10.1 GB |
179
+
180
+ The 8B matrices expect 4096 input dimensions instead of 2560; the node checks the width and refuses a mismatch.
181
+
182
+ ---
183
+
184
+ ## The benchmark
185
+
186
+ Nothing here is a single render. Every figure comes from **three seeds — 42, 100 000 and 100 000 000** — chosen far apart so no one can suspect they are correlated.
187
+
188
+ | | volume |
189
+ |---|---|
190
+ | Speech | 297 renders — 9 conditionings × 11 languages × 3 seeds |
191
+ | Image | 405 renders — 9 conditionings × 15 prompts × 3 seeds |
192
+
193
+ **Speech** is scored in phonemes, by [ZIPA-CR-large](https://huggingface.co/anyspeech/zipa-large-crctc-ns-800k) (88 languages, no lexical decoder — it will not silently repair a botched syllable into a real word), against the 32B **of the same seed**. Distances are Levenshtein throughout.
194
+
195
+ **Image** is scored by SigLIP2 so400m, both against the 32B's render and against the prompt itself — on that second axis the 32B is just one column among nine.
196
+
197
+ Languages: en, fr, es, de, it, pt, ru, ar, zh, ja, ko.
198
+
199
+ ---
200
+
201
+ ## Results
202
+
203
+ | Conditioning | Speech | ± | Image | Prompt |
204
+ |---|---|---|---|---|
205
+ | **32B** *(reference)* | **100.0** | — | **100.0** | **100.0** |
206
+ | `8b-ClipProj-v3.1` | **98.8** | ±1.4 | 100.0 | 100.4 |
207
+ | `8b-ClipProj-v3.1-mlp` | 98.2 | ±1.2 | 101.4 | **102.3** |
208
+ | `4b-ClipProj-v3.1` | 97.9 | ±1.0 | 100.1 | 99.4 |
209
+ | `4b-ClipProj-v3.1-mlp` | 97.8 | ±1.1 | 100.9 | 100.3 |
210
+ | *v3 ridge / mlp, 4B and 8B* | *93.2 – 95.6* | | *99.4 – 102.4* | *99.0 – 100.7* |
211
+
212
+ `±` is the spread of the score across the three seeds. **Two models separated by less than that are not separated at all.**
213
+
214
+ ### The raw counts behind the speech score
215
+
216
+ Three metrics, three units, never added together. **PER** counts phonemes over three seeds on the current
217
+ protocol; **WER** and **CER** count words and characters as Whisper hears them, single-seed on the earlier
218
+ 0.3 MP protocol. They are listed side by side because they disagree in useful ways — see the language
219
+ breakdown below.
220
+
221
+ | | **PER** (3 seeds) | | **WER** (1 seed) | **CER** (1 seed) |
222
+ |---|---|---|---|---|
223
+ | **32B** *(reference)* | **0 / 2469** | — | 6 / 174 | 6 / 869 |
224
+ | *32B against itself* | *~193 / 2469* | *7.8 %* | — | — |
225
+ | `8b-v3.1` | **211 / 2469** | 8.5 % | **10 / 174** | **14 / 869** |
226
+ | `8b-v3.1-mlp` | 222 / 2469 | 9.0 % | 15 / 174 | 28 / 869 |
227
+ | `4b-v3.1-mlp` | 230 / 2469 | 9.3 % | 13 / 174 | 23 / 869 |
228
+ | `4b-v3.1` | 232 / 2469 | 9.4 % | 15 / 174 | 30 / 869 |
229
+ | `4b-v3-mlp` | 281 / 2469 | 11.4 % | 18 / 174 | 32 / 869 |
230
+ | `8b-v3-mlp` | 317 / 2469 | 12.8 % | 17 / 174 | 29 / 869 |
231
+ | `8b-v3` | 326 / 2469 | 13.2 % | 23 / 174 | 46 / 869 |
232
+ | `4b-v3` | 341 / 2469 | 13.8 % | 19 / 174 | 36 / 869 |
233
+
234
+ The 32B scores 0 on PER by construction — it *is* the reference. The row below it is the meaningful one:
235
+ compared to **itself** on another seed it drifts by about 7.8 %, and the four v3.1 files sit at 8.5 to
236
+ 9.4 %. The v3 files sit at 11.4 to 13.8 %, clear of that band.
237
+
238
+ WER and CER rank the files in nearly the same order, which is the point of quoting both: `8b-v3.1` leads
239
+ all three metrics, and no v3 file beats any v3.1 file on any of them.
240
+
241
+ ### Why some scores exceed 100 — and why that is not "better than the 32B"
242
+
243
+ The two image columns do not share a reference, and neither exceedance means what it looks like.
244
+
245
+ **Prompt.** This axis is `cos(image embedding, prompt embedding)`. The 32B is **not** the reference here —
246
+ it is one column among nine, and its value is set to 100 only to give the scale a fixed point. Nothing
247
+ requires it to be the best, and it demonstrably is not: on the prompt asking for a loaf **cut in two**, it
248
+ renders a single piece on two seeds out of three. A projection that follows the description more closely
249
+ earns a higher cosine, legitimately.
250
+
251
+ **Image.** Here 100 *is* the 32B against itself, but the comparison is asymmetric: the threshold pits
252
+ `32B(seed 42)` against `32B(seed 7391)` — two different draws — while a projection is compared to
253
+ `32B(seed 42)`, the **same** draw. It plays with its reference's seed, so the draw noise is removed on its
254
+ side. A score of 101.4 says only *closer to that 32B render than two 32B renders are to each other*.
255
+
256
+ **And none of it is significant.** The nine models span 0.0048 of cosine on the prompt axis, against a
257
+ within-model standard deviation of 0.024 to 0.029 — five times larger. The paired test over 45 cases calls
258
+ all eight projections indistinguishable from the 32B, including the one at 102.3. The +2.3 % is real as a
259
+ measurement and void as a result.
260
+
261
+ ### What actually separates
262
+
263
+ **The corpus, not the size and not the head.** All four v3.1 land within one point of each other — 97.8 to 98.8 — while ranging from 4.6 to 10.1 GB. All four v3 sit a clear notch below, 93.2 to 95.6, at identical sizes. A 4B v3.1 beats an 8B v3 by four points while weighing half as much.
264
+
265
+ **Nothing separates in image.** All nine conditionings, v3 included, are at or above the threshold: 99.4 to 102.4. Swapping the 32B for a 4B changes the picture **less than changing the seed does**. On this axis the 32B is not a ceiling — `8b-v3.1-mlp` scores 102.3 for prompt fidelity, and on one prompt asking for a loaf cut in two, the 32B rendered a single piece on two seeds out of three while the 8B ridge rendered two on all three.
266
+
267
+ **4B against 8B does not separate on general pronunciation.** 97.8 against 98.8, for a seed-to-seed spread of ±1.0 to ±1.4. If you need one number: they are the same.
268
+
269
+ **Ridge against MLP does not separate either.** The residual buys nothing measurable in speech. It shows up in image prompt fidelity — 102.3 against 100.4 on the 8B — but that axis has its own noise and I would not choose a file on it.
270
+
271
+ ### The image measurements in full
272
+
273
+ Two independent SigLIP2 so400m readings over the same 405 renders. First, resemblance to the 32B's own render, same prompt and same seed — 45 cases per model:
274
+
275
+ | | cosine | std. dev. | % of threshold | worst prompt |
276
+ |---|---|---|---|---|
277
+ | **32B against itself** | **0.9552** | — | **100.0** | — |
278
+ | `8b-v3-mlp` | 0.9653 | 0.0328 | 102.4 | 0.8804 |
279
+ | `8b-v3.1-mlp` | 0.9612 | 0.0363 | 101.4 | 0.8808 |
280
+ | `8b-v3` | 0.9594 | 0.0350 | 101.0 | 0.8910 |
281
+ | `4b-v3.1-mlp` | 0.9590 | 0.0321 | 100.9 | 0.9038 |
282
+ | `4b-v3-mlp` | 0.9588 | 0.0408 | 100.8 | 0.8989 |
283
+ | `4b-v3.1` | 0.9557 | 0.0445 | 100.1 | 0.8766 |
284
+ | `8b-v3.1` | 0.9552 | 0.0448 | 100.0 | 0.8893 |
285
+ | `4b-v3` | 0.9528 | 0.0423 | 99.4 | 0.8861 |
286
+
287
+ *Floor: 0.5313 — two 32B renders sharing no content at all still score that, on style and generator artefacts alone.*
288
+
289
+ **The standard deviation settles it.** It runs 0.032 to 0.045, while the entire spread from best to worst model is 0.0125. The scatter within one model is three to four times the gap between models. Nothing here is a ranking.
290
+
291
+ Second, fidelity to the written prompt — an axis where the 32B is one column among nine rather than the reference:
292
+
293
+ | | cosine | std. dev. | base 100 |
294
+ |---|---|---|---|
295
+ | `8b-v3.1-mlp` | 0.1504 | 0.0252 | **102.3** |
296
+ | `4b-v3-mlp` | 0.1481 | 0.0275 | 100.7 |
297
+ | `8b-v3.1` | 0.1477 | 0.0290 | 100.4 |
298
+ | `4b-v3.1-mlp` | 0.1475 | 0.0249 | 100.3 |
299
+ | **32B** | 0.1471 | 0.0245 | **100.0** |
300
+ | `4b-v3.1` | 0.1462 | 0.0272 | 99.4 |
301
+ | `8b-v3` | 0.1460 | 0.0267 | 99.3 |
302
+ | `8b-v3-mlp` | 0.1458 | 0.0236 | 99.2 |
303
+ | `4b-v3` | 0.1456 | 0.0268 | 99.0 |
304
+
305
+ *Floor: −0.0293 — one scene's image against another scene's prompt.*
306
+
307
+ Same verdict, and harder: the spread across all nine models is 0.0048 for a standard deviation of 0.024 to 0.029, **five times larger**. Four models sit above the 32B and four below, in an order that carries no information.
308
+
309
+ The two image axes do not even agree with each other: `8b-v3-mlp` tops the resemblance table and sits second from last on prompt fidelity. Imitating the 32B and following the prompt are not the same objective — the 32B itself misses prompts.
310
+
311
+ ---
312
+
313
+ ## What counts as an error
314
+
315
+ One error is **one phoneme inserted, deleted or substituted** relative to what the 32B pronounced — same prompt, same seed. Levenshtein distance, nothing weighted, nothing forgiven.
316
+
317
+ There is no dictionary in the loop. ZIPA transcribes sound to IPA and has no lexical decoder, so it will not quietly repair a botched syllable into a real word the way a speech-to-text engine would. What it writes down is what came out of the speaker.
318
+
319
+ Concretely, on the French line *"la lumière de Marseille"*, seed 42 — the phonemes following `d ɛ` ("de"):
320
+
321
+ | | | heard as | errors on the line |
322
+ |---|---|---|---|
323
+ | **32B** | `m a ʀ s ɛ j` | *Marseille* | — |
324
+ | `8b-v3.1` | `m a ʀ s ɛ j` | *Marseille* | 4 / 71 |
325
+ | `4b-v3.1` | `m a ʀ s ɛ ʀ ɛ` | *"marcerre"* | 5 / 71 |
326
+ | `4b-v3` | `m a z ɛ ʀ` | *"mazer"* | 8 / 71 |
327
+
328
+ Note how little the toponym costs: `4b-v3.1` botches the name outright and pays **one** phoneme more than `8b-v3.1` over the whole sentence. That is exactly why the aggregate scores cannot settle the proper-noun question, and why it gets its own section below rather than a place in the ranking.
329
+
330
+ ## Per language, because the average hides everything
331
+
332
+ Errors are counted against the 32B **of the same seed**, averaged over the three seeds. The first two columns are the yardstick: how long the reference is, and how much the 32B differs from *itself*.
333
+
334
+ | | length | **threshold** | `8b-v3.1` | `8b-v3.1-mlp` | `4b-v3.1` | `4b-v3.1-mlp` |
335
+ |---|---|---|---|---|---|---|
336
+ | en | 77 | **0.0** | 0.7 | 0.7 | 0.7 | 0.0 |
337
+ | es | 62 | **0.0** | 1.3 | 0.0 | 0.7 | 0.0 |
338
+ | de | 97 | 2.7 | 3.0 | 5.0 | 3.7 | 2.7 |
339
+ | it | 62 | 2.3 | 0.7 | 0.3 | 4.3 | 1.0 |
340
+ | zh | 85 | 3.3 | 3.3 | 1.0 | 2.7 | 2.0 |
341
+ | fr | 70 | 4.0 | 1.7 | 1.3 | 5.0 | 4.0 |
342
+ | ar | 94 | 6.3 | 5.7 | 4.7 | 7.7 | 12.3 |
343
+ | ja | 73 | 6.7 | 6.3 | 6.7 | 5.7 | 7.0 |
344
+ | ru | 77 | 10.7 | 14.0 | 16.3 | 14.0 | 10.0 |
345
+ | ko | 64 | 11.3 | 11.0 | 15.0 | 10.7 | 9.3 |
346
+ | pt | 62 | **17.0** | 22.7 | 23.0 | 22.3 | 28.3 |
347
+
348
+ Read it against the threshold column, never in absolute terms:
349
+
350
+ - **English and Spanish** — the 32B repeats itself phoneme for phoneme. There, a single phoneme of drift is real signal, and all four files stay within one.
351
+ - **French, Italian, Chinese, Arabic, Japanese** — the projections are *at or below* the 32B's own variance. In French both 8B files land at 1.7 and 1.3 against a threshold of 4.0: closer to the 32B than the 32B is to itself.
352
+ - **Portuguese, Russian and Korean** carry thresholds of 17.0, 10.7 and 11.3 — the reference rewrites a large share of its own pronunciation between seeds. Any single-seed comparison there was measuring the dice.
353
+
354
+ ### Where the phoneme metric misleads, and the cross-check that catches it
355
+
356
+ A high threshold does not mean the speech is bad. It means **the phoneme transcriber cannot hold that
357
+ language still.** Cross-checking against Whisper, which reads words rather than sounds, on the same
358
+ renders:
359
+
360
+ | | ZIPA threshold | ZIPA v3.1 | × threshold | **Whisper, 32B** | **Whisper, v3.1** |
361
+ |---|---|---|---|---|---|
362
+ | pt | 17.0 | 24.1 | **1.42** | **0 / 88** | **1.8 / 88** |
363
+ | ru | 10.7 | 13.6 | 1.27 | **0 / 85** | 2.8 / 85 |
364
+ | ko | 11.3 | 11.5 | 1.01 | 2 / 85 | 3.2 / 85 |
365
+ | ja | 6.7 | 6.4 | 0.96 | 2 / 39 | 5.2 / 39 |
366
+ | zh | 3.3 | 2.2 | 0.67 | 2 / 31 | 3.5 / 31 |
367
+
368
+ **Portuguese is the worst language by phoneme and one of the best by word** — zero character errors for
369
+ the 32B, 2 % for the v3.1 files. Russian likewise: Whisper transcribes the 32B and two of the projections
370
+ word for word.
371
+
372
+ The cause is exactly what makes ZIPA useful elsewhere: it has no lexical decoder. European Portuguese
373
+ elides and reduces its vowels, Russian has vowel reduction under stress shift — the phonetic realisation
374
+ moves from one draw to the next while the word does not. ZIPA counts every allophonic variation as an
375
+ error; Whisper, which recognises the word, sees none. In Japanese and Chinese the bias runs the other
376
+ way: Whisper is harsher, because one missed ideogram weighs heavily on 31 characters.
377
+
378
+ **Neither metric is sufficient alone.** Where the ZIPA threshold is high, read the word column.
379
+ (Whisper figures are single-seed, on the earlier 0.3 MP protocol.)
380
+
381
+ ---
382
+
383
+ ## The 32B is one of the least stable models here
384
+
385
+ Distance between two renders of the **same** model, seed changed, nothing else:
386
+
387
+ | | against itself | against the 32B |
388
+ |---|---|---|
389
+ | `4b-v3.1` | **3.6** | 7.0 |
390
+ | `4b-v3.1-mlp` | 4.0 | 7.0 |
391
+ | `8b-v3.1` | 4.1 | 6.4 |
392
+ | `8b-v3.1-mlp` | 4.7 | 6.7 |
393
+ | **32B** | **5.8** | — |
394
+
395
+ The projections repeat themselves *better* than the model they imitate.
396
+
397
+ **And the gap to the 32B is reproducible, not random.** Each projection sits far closer to itself (3.6–4.7) than to the 32B (6.4–7.0). If swapping the encoder merely added randomness, those two columns would match. They do not — each file redoes the same offset on every seed.
398
+
399
+ **It is not an accent either.** An accent would mean one phoneme consistently rendered as another. Counting the actual substitutions says otherwise:
400
+
401
+ | | substitutions | covered by recurring patterns |
402
+ |---|---|---|
403
+ | **32B against itself** | **103** | ɑ→a ×10, ɾ→r ×6, ʒ→ʐ ×5 |
404
+ | `8b-v3.1` | **103** | 4 % — one pattern |
405
+ | `4b-v3.1` | 114 | **0 %** |
406
+ | `8b-v3.1-mlp` | 115 | 8 % |
407
+ | `4b-v3.1-mlp` | 125 | **0 %** |
408
+ | the four v3 files | 153–196 | 2–13 % |
409
+
410
+ The v3.1 files produce **as many substitutions as the 32B inflicts on itself** — 103 to 125 against 103 — and almost none of them form a repeating pattern. The offset is reproducible but scattered across many different sounds rather than concentrated into a signature. Ironically the clearest patterns belong to the 32B itself, between its own seeds, where they are ordinary allophonic variation.
411
+
412
+ The v3 files produce 1.5 to 2 times as many.
413
+
414
+ None of which tells you what it *sounds* like. A native speaker might well hear something none of these counts describe.
415
+
416
+ ---
417
+
418
+ ## One word, one language
419
+
420
+ The French prompt says *"la lumière de **Marseille**"*. Six phonemes out of seventy — the error rate drowns them, the ear does not:
421
+
422
+ | | seed 42 | seed 100 k | seed 100 M |
423
+ |---|---|---|---|
424
+ | 32B | ✓ | ✓ | ✓ |
425
+ | `8b-v3.1` | ✓ | ✓ | ✓ |
426
+ | `8b-v3.1-mlp` | ✓ | ✓ | ✓ |
427
+ | `4b-v3.1-mlp` | ✗ | ✗ | ✓ |
428
+ | `4b-v3.1` | ✗ | ✗ | ✗ |
429
+
430
+ Treat this for what it is: **one proper noun, in one of eleven languages, on three seeds.** It is not a ranking criterion and the aggregate scores already say 4B and 8B are equivalent. It is a hint that the two sizes may diverge on rare lexical items even where they agree on everything else — consistent with quantisation costing facts, which was already known. If your prompts lean on proper nouns, test both before deciding.
431
+
432
+ ---
433
+
434
+ ## Which one to take
435
+
436
+ | If you | Take |
437
+ |---|---|
438
+ | generate images or video without speech | **`4b-ClipProj-v3.1`** — 4.6 GB, indistinguishable from the 32B |
439
+ | are tight on VRAM | **`4b-ClipProj-v3.1`** — the ridge is 26 MB and gives up nothing measurable |
440
+ | generate multilingual speech | **`8b-ClipProj-v3.1`** — best speech score, and steady on proper nouns |
441
+ | rely on named people or places | **`8b-ClipProj-v3.1`**, and run the 32B once to check the name works there at all |
442
+
443
+ Do not take a v3: it is the only difference this benchmark resolves cleanly.
444
+
445
+ ---
446
+
447
+ ## How the speech benchmark got affordable
448
+
449
+ The old protocol rendered full 0.3 MP video and threw the picture away. Measured, same prompt and seed:
450
+
451
+ | | time |
452
+ |---|---|
453
+ | 0.3 MP video + audio + previews *(old)* | 80.3 s |
454
+ | same, video VAE and previews removed | 48.7 s |
455
+ | **64×64, 8 steps, audio only** | **12.4 s** |
456
+
457
+ Verified lossless before adopting: **+0.2 dB** across every octave band and **1 phoneme out of 70** for dropping the video decode; 64×64 costs 3 phonemes out of 70 against the full-resolution render.
458
+
459
+ **128×128 was rejected** — it truncates the start of the sentence, exactly the same eight phonemes at 6 steps and at 8. 64×64 does not. Counter-intuitive, reproducible, and the reason the whole benchmark runs at the smaller size.
460
+
461
+ That is what made three seeds across 297 renders possible at all: 31 minutes on two cards instead of six and a half hours.
462
+
463
+ ---
464
+
465
+ ## Limitations
466
+
467
+ **Three seeds fix the order of magnitude of the noise, not its tail.** Any gap under one point of score is not a result.
468
+
469
+ **The cosine is blind to countable attributes.** A whole loaf and a halved loaf, same crust, same paper, same light, give the same vector to the fourth decimal. Image equivalence here means *global appearance*, not attribute-by-attribute conformity.
470
+
471
+ **Speech quality is deliberately poor.** Six to eight steps gives a tinny, canned sound — identically for the 32B, with the same 19 dB dip between 1 and 3 kHz. The benchmark measures **correctness of pronunciation, not fidelity of reproduction**.
472
+
473
+ **The phoneme metric is unreliable in Portuguese, Russian and Korean** — not the speech itself. The
474
+ reference drifts by 17.0, 10.7 and 11.3 phonemes there between seeds, while Whisper transcribes the same
475
+ renders with zero to three character errors. Read the word column in those languages.
476
+
477
+ **Quantisation costs facts.** Known before, still true, and the most likely explanation for the proper-noun gap.
478
+
479
+ ---
480
+
481
+ ## Licence and responsibility
482
+
483
+ MIT, like the node. These matrices are derived from the activations of both models and their legal status is unclear; they are provided as-is, for research.
484
+
485
+ - **Qwen3-VL** — Alibaba, Apache 2.0.
486
+ - **MiniMax H3** — custom licence, read it before any commercial use.
487
+
488
+ Not affiliated with, endorsed by, or connected to Alibaba / Qwen, MiniMax, or Comfy Org. You remain responsible for what you generate.
489
+
490
+ ---
491
+
492
+ ## Credits
493
+
494
+ Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number here was measured on this hardware, never estimated. Where a prediction lost to a measurement, the measurement won and the text was rewritten — which happened three times in this release, the largest being a single-seed ranking of the v3.1 files that dissolved entirely once the threshold was known.
bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e53d45ea1012723b3857fb14471a40455538254236edac1b437bf017d40fdffd
3
+ size 429755
bench3.1/audio/ru/4b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14b85ea705b93d1e4c0410682699b17bce06c6a0bd32415ee3515e86f88a67b8
3
+ size 453961
bench3.1/audio/ru/4b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:29607b7fa47691dbbc453ce9122db5978aaa8b4263f4bf757e2df21f915e5f7d
3
+ size 458935
bench3.1/audio/ru/4b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:10b17ba27faed991a156909765e22fb28d79d838bc2f01f8c1f90bd5c5c358ca
3
+ size 424544
bench3.1/audio/ru/4b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b915e71b290af104f38fd9123e4642c82673b1f59c09cc156462bb847a25515c
3
+ size 447572
bench3.1/audio/ru/4b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b666c951a80e641a69dd57a7c13c0f64a6638c8b108b47e7ef3e6029ee5c38dc
3
+ size 444602
bench3.1/audio/ru/4b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fea28c5ba8d35e83900d7849288250acd27bfe3e2e53578e46e2012421102fc1
3
+ size 423455
bench3.1/audio/ru/4b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:96eb417bd32b0b5e41e1df444bef5eb703894705df4271f7582b0aaa5c36388e
3
+ size 456929
bench3.1/audio/ru/8b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:70616fe97638cadc8c565e8b9be5d6f19eb12b14f716c71d0f4d7076f3848694
3
+ size 470866
bench3.1/audio/ru/8b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ce3fed11e203360f22629ec594545bf49fc358353649a7c9355c722e7c80fd0
3
+ size 430723
bench3.1/audio/ru/8b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:09c5943d5941fad3be131442fa49686347df70a1c8cdefbdf605f6897a8ae8ec
3
+ size 460268
bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3bb419c0f71433d86a0ebd5acd9ca657c62bdcea415c41a196c53eab887136fb
3
+ size 468965
bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82c5603addbf88dad2286e88ab9934b21d7c1d1a6e2749f30b1c5692ce24b198
3
+ size 435985
bench3.1/audio/ru/8b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:45307a209ab81d094164ca22ae0918b08ae5e7c7ccd0d0783273c814786a68b2
3
+ size 467116
bench3.1/audio/ru/8b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e2f5e8d57020673cb2a886dc91ba798020d5da4dcc15ebcd6660ebb8f1870b4c
3
+ size 468444
bench3.1/audio/ru/8b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a6257d388f6c2e52877175e2772690649683229daddfcd9181237b0e2905b4f9
3
+ size 435118
bench3.1/audio/ru/8b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a42b305503a82ae3a99f41a1af9d8310ef2ee69fe4c940c0a7b57e80b6f3b306
3
+ size 451851
bench3.1/audio/ru/8b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7c518a115cfbc962d60b7c37be08b265f158413581d7772752e468fc2a6116b1
3
+ size 443896
bench3.1/audio/ru/8b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cc08c210a498207db38c69a7559f68dfbe35ceba9432a8dddaadd936822bd7d
3
+ size 429043
bench3.1/audio/ru/8b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8a0c109adde39f7232df81be57dcfc0121469808e2080e7598ad5c9fe47524c
3
+ size 468524
bench3.1/audio/zh/32b_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e599bba7269aa0ff769c08cf9c77ef0bb0bf2ee2496a362999a1a5de7e0f8cf0
3
+ size 461955
bench3.1/audio/zh/32b_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e35810382807ae12b022cbad36c09b6cc8a33f91b4e749436db64a8f4157ca82
3
+ size 398314
bench3.1/audio/zh/32b_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1b5a53b427844cf39df6ba6081ea9ca360e16c39bb2ed0fea2d3d6391d647f77
3
+ size 427973
bench3.1/audio/zh/4b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d8b8f18e68a84190fed01c48343c85cbf678846293edc9bd436075b83b6ddd4
3
+ size 398882
bench3.1/audio/zh/4b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dbf70317d461f64035ad8bf0105a139733a23bd7285034b2d9a11ef0b3fee808
3
+ size 377409
bench3.1/audio/zh/4b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:903757ce6ea45aba792380893263e876a3234921c625568d01376503c5beb9d1
3
+ size 371468
bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1fc3444c32f70d3eedbf26d2568646fb9db7a3cd5a7e83ed99780b0b86306718
3
+ size 413401
bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54b63f5686a68118ebe145e05f99cb7c6d489ca9d1991bcd9aa805c3cdef8e60
3
+ size 373573
bench3.1/audio/zh/4b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff260b95a903f8de90b8c01c0b9c6ae4dbb60b7b6a9f8890bae01d8e6a202523
3
+ size 382691
bench3.1/audio/zh/4b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:76d6d0f72e94c95546165d5f573e157fc6d1a23dba9bd1cccc2d748c4e95eb98
3
+ size 389361
bench3.1/audio/zh/4b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f4110d6ee92c00ef0b480dc971e41b47a9bb6dc6140f8be8b90c1704056f402
3
+ size 375759
bench3.1/audio/zh/4b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a7d4d3dfc432e1d83c2ed12dbe8f1ad98e7e33e30a8c643431898db5368e7492
3
+ size 384508
bench3.1/audio/zh/4b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c949520895168b02258ef3b15960d39c40630dd77f9a0f5d8454558c1e06d364
3
+ size 378740
bench3.1/audio/zh/4b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c2785d72b502af9ce29666f640a6edc3afc09d47acf998c74351104d74ab01b0
3
+ size 379253
bench3.1/audio/zh/4b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e524606a0168948770a5dbb457c4aae58f1c26261308b627f789efd39b7d555f
3
+ size 376275
bench3.1/audio/zh/8b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1afd8836040c25f5198e5f6e682b2587dedb9cbcf91e7dcf3e1922995975154a
3
+ size 401529
bench3.1/audio/zh/8b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ba18aea41c4b90f86b2508c3c27034466dd872c9e77247904fec8d45fb5613a6
3
+ size 364176
bench3.1/audio/zh/8b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6eaddc0f3ad47127c8c58e59c1c43b50103d6bb11807a681dcbecd20c4f05cb2
3
+ size 372752
bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ebf2e43d6246efb00c0bcb951e796ef32a9cd98db9c5efe27a7bbb993354f4b8
3
+ size 413868
bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8bb615863a72acfe1a15daa451c0278374ca40f8c66982a1d081d9ea1b57c59d
3
+ size 393327
bench3.1/audio/zh/8b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dc76d98d90a4b637f591317a810d9ad36a5b8b3d78d83a72a07dccf61f659514
3
+ size 387031
bench3.1/audio/zh/8b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12791b0ea5979de46cdebb38c26d0411b6fc54638c4067f2d2a2082d8c6a3042
3
+ size 385676
bench3.1/audio/zh/8b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a6bb13e971f054676dd2cded560927fea9053a53dee5fc9f1713796921277485
3
+ size 365478
bench3.1/audio/zh/8b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:190bd4685e28a1188e8582ad09b0923ce72b8f9f3b15aa4b6a4f228d94bbc219
3
+ size 372457
bench3.1/audio/zh/8b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f89932a5e632580703ac9c3214c964eba4b33d71fb5103934642c7bd46c3bc6d
3
+ size 370772
bench3.1/audio/zh/8b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8823243f92f9310e50fa6565b5b7df61d104ce18d178967a8f7a9cde373d8681
3
+ size 360986
bench3.1/audio/zh/8b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:229f7ea3267d1568e0f4051d8e56604ebcc48c08d13c52c007b0d905cbe5dcc3
3
+ size 369764