misstoyou NicoLab28 commited on
Commit
6111b26
·
0 Parent(s):

Duplicate from NicoLab28/ClipProj-MiniMax-H3

Browse files

Co-authored-by: Lab <NicoLab28@users.noreply.huggingface.co>

This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +527 -0
  2. README.md +234 -0
  3. RELEASE_NOTES_v3.md +160 -0
  4. bench3.1/README.md +506 -0
  5. bench3.1/audio/ar/32b_s100000.flac +3 -0
  6. bench3.1/audio/ar/32b_s100000000.flac +3 -0
  7. bench3.1/audio/ar/32b_s42.flac +3 -0
  8. bench3.1/audio/ar/4b-v3-mlp_s100000.flac +3 -0
  9. bench3.1/audio/ar/4b-v3-mlp_s100000000.flac +3 -0
  10. bench3.1/audio/ar/4b-v3-mlp_s42.flac +3 -0
  11. bench3.1/audio/ar/4b-v3.1-mlp_s100000.flac +3 -0
  12. bench3.1/audio/ar/4b-v3.1-mlp_s100000000.flac +3 -0
  13. bench3.1/audio/ar/4b-v3.1-mlp_s42.flac +3 -0
  14. bench3.1/audio/ar/4b-v3.1_s100000.flac +3 -0
  15. bench3.1/audio/ar/4b-v3.1_s100000000.flac +3 -0
  16. bench3.1/audio/ar/4b-v3.1_s42.flac +3 -0
  17. bench3.1/audio/ar/4b-v3_s100000.flac +3 -0
  18. bench3.1/audio/ar/4b-v3_s100000000.flac +3 -0
  19. bench3.1/audio/ar/4b-v3_s42.flac +3 -0
  20. bench3.1/audio/ar/8b-v3-mlp_s100000.flac +3 -0
  21. bench3.1/audio/ar/8b-v3-mlp_s100000000.flac +3 -0
  22. bench3.1/audio/ar/8b-v3-mlp_s42.flac +3 -0
  23. bench3.1/audio/ar/8b-v3.1-mlp_s100000.flac +3 -0
  24. bench3.1/audio/ar/8b-v3.1-mlp_s100000000.flac +3 -0
  25. bench3.1/audio/ar/8b-v3.1-mlp_s42.flac +3 -0
  26. bench3.1/audio/ar/8b-v3.1_s100000.flac +3 -0
  27. bench3.1/audio/ar/8b-v3.1_s100000000.flac +3 -0
  28. bench3.1/audio/ar/8b-v3.1_s42.flac +3 -0
  29. bench3.1/audio/ar/8b-v3_s100000.flac +3 -0
  30. bench3.1/audio/ar/8b-v3_s100000000.flac +3 -0
  31. bench3.1/audio/ar/8b-v3_s42.flac +3 -0
  32. bench3.1/audio/de/32b_s100000.flac +3 -0
  33. bench3.1/audio/de/32b_s100000000.flac +3 -0
  34. bench3.1/audio/de/32b_s42.flac +3 -0
  35. bench3.1/audio/de/4b-v3-mlp_s100000.flac +3 -0
  36. bench3.1/audio/de/4b-v3-mlp_s100000000.flac +3 -0
  37. bench3.1/audio/de/4b-v3-mlp_s42.flac +3 -0
  38. bench3.1/audio/de/4b-v3.1-mlp_s100000.flac +3 -0
  39. bench3.1/audio/de/4b-v3.1-mlp_s100000000.flac +3 -0
  40. bench3.1/audio/de/4b-v3.1-mlp_s42.flac +3 -0
  41. bench3.1/audio/de/4b-v3.1_s100000.flac +3 -0
  42. bench3.1/audio/de/4b-v3.1_s100000000.flac +3 -0
  43. bench3.1/audio/de/4b-v3.1_s42.flac +3 -0
  44. bench3.1/audio/de/4b-v3_s100000.flac +3 -0
  45. bench3.1/audio/de/4b-v3_s100000000.flac +3 -0
  46. bench3.1/audio/de/4b-v3_s42.flac +3 -0
  47. bench3.1/audio/de/8b-v3-mlp_s100000.flac +3 -0
  48. bench3.1/audio/de/8b-v3-mlp_s100000000.flac +3 -0
  49. bench3.1/audio/de/8b-v3-mlp_s42.flac +3 -0
  50. bench3.1/audio/de/8b-v3.1-mlp_s100000.flac +3 -0
.gitattributes ADDED
@@ -0,0 +1,527 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ demo/chess-32b-reference.mp4 filter=lfs diff=lfs merge=lfs -text
37
+ demo/chess-4b-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
38
+ demo/chess-4b-ridge.mp4 filter=lfs diff=lfs merge=lfs -text
39
+ demo/chess-8b-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
40
+ demo/chess-8b-ridge.mp4 filter=lfs diff=lfs merge=lfs -text
41
+ demo/chess-comparison.mp4 filter=lfs diff=lfs merge=lfs -text
42
+ bench3.1/audio/ar/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
43
+ bench3.1/audio/ar/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
44
+ bench3.1/audio/ar/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
45
+ bench3.1/audio/ar/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
46
+ bench3.1/audio/ar/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
47
+ bench3.1/audio/ar/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
48
+ bench3.1/audio/ar/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
49
+ bench3.1/audio/ar/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
50
+ bench3.1/audio/ar/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
51
+ bench3.1/audio/ar/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
52
+ bench3.1/audio/ar/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
53
+ bench3.1/audio/ar/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
54
+ bench3.1/audio/ar/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
55
+ bench3.1/audio/ar/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
56
+ bench3.1/audio/ar/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
57
+ bench3.1/audio/ar/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
58
+ bench3.1/audio/ar/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
59
+ bench3.1/audio/ar/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
60
+ bench3.1/audio/ar/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
61
+ bench3.1/audio/ar/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
62
+ bench3.1/audio/ar/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
63
+ bench3.1/audio/ar/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
64
+ bench3.1/audio/ar/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
65
+ bench3.1/audio/ar/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
66
+ bench3.1/audio/ar/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
67
+ bench3.1/audio/ar/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
68
+ bench3.1/audio/ar/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
69
+ bench3.1/audio/de/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
70
+ bench3.1/audio/de/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
71
+ bench3.1/audio/de/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
72
+ bench3.1/audio/de/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
73
+ bench3.1/audio/de/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
74
+ bench3.1/audio/de/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
75
+ bench3.1/audio/de/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
76
+ bench3.1/audio/de/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
77
+ bench3.1/audio/de/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
78
+ bench3.1/audio/de/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
79
+ bench3.1/audio/de/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
80
+ bench3.1/audio/de/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
81
+ bench3.1/audio/de/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
82
+ bench3.1/audio/de/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
83
+ bench3.1/audio/de/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
84
+ bench3.1/audio/de/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
85
+ bench3.1/audio/de/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
86
+ bench3.1/audio/de/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
87
+ bench3.1/audio/de/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
88
+ bench3.1/audio/de/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
89
+ bench3.1/audio/de/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
90
+ bench3.1/audio/de/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
91
+ bench3.1/audio/de/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
92
+ bench3.1/audio/de/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
93
+ bench3.1/audio/de/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
94
+ bench3.1/audio/de/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
95
+ bench3.1/audio/de/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
96
+ bench3.1/audio/en/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
97
+ bench3.1/audio/en/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
98
+ bench3.1/audio/en/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
99
+ bench3.1/audio/en/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
100
+ bench3.1/audio/en/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
101
+ bench3.1/audio/en/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
102
+ bench3.1/audio/en/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
103
+ bench3.1/audio/en/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
104
+ bench3.1/audio/en/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
105
+ bench3.1/audio/en/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
106
+ bench3.1/audio/en/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
107
+ bench3.1/audio/en/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
108
+ bench3.1/audio/en/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
109
+ bench3.1/audio/en/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
110
+ bench3.1/audio/en/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
111
+ bench3.1/audio/en/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
112
+ bench3.1/audio/en/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
113
+ bench3.1/audio/en/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
114
+ bench3.1/audio/en/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
115
+ bench3.1/audio/en/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
116
+ bench3.1/audio/en/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
117
+ bench3.1/audio/en/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
118
+ bench3.1/audio/en/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
119
+ bench3.1/audio/en/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
120
+ bench3.1/audio/en/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
121
+ bench3.1/audio/en/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
122
+ bench3.1/audio/en/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
123
+ bench3.1/audio/es/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
124
+ bench3.1/audio/es/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
125
+ bench3.1/audio/es/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
126
+ bench3.1/audio/es/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
127
+ bench3.1/audio/es/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
128
+ bench3.1/audio/es/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
129
+ bench3.1/audio/es/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
130
+ bench3.1/audio/es/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
131
+ bench3.1/audio/es/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
132
+ bench3.1/audio/es/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
133
+ bench3.1/audio/es/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
134
+ bench3.1/audio/es/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
135
+ bench3.1/audio/es/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
136
+ bench3.1/audio/es/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
137
+ bench3.1/audio/es/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
138
+ bench3.1/audio/es/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
139
+ bench3.1/audio/es/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
140
+ bench3.1/audio/es/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
141
+ bench3.1/audio/es/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
142
+ bench3.1/audio/es/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
143
+ bench3.1/audio/es/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
144
+ bench3.1/audio/es/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
145
+ bench3.1/audio/es/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
146
+ bench3.1/audio/es/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
147
+ bench3.1/audio/es/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
148
+ bench3.1/audio/es/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
149
+ bench3.1/audio/es/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
150
+ bench3.1/audio/fr/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
151
+ bench3.1/audio/fr/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
152
+ bench3.1/audio/fr/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
153
+ bench3.1/audio/fr/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
154
+ bench3.1/audio/fr/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
155
+ bench3.1/audio/fr/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
156
+ bench3.1/audio/fr/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
157
+ bench3.1/audio/fr/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
158
+ bench3.1/audio/fr/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
159
+ bench3.1/audio/fr/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
160
+ bench3.1/audio/fr/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
161
+ bench3.1/audio/fr/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
162
+ bench3.1/audio/fr/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
163
+ bench3.1/audio/fr/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
164
+ bench3.1/audio/fr/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
165
+ bench3.1/audio/fr/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
166
+ bench3.1/audio/fr/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
167
+ bench3.1/audio/fr/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
168
+ bench3.1/audio/fr/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
169
+ bench3.1/audio/fr/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
170
+ bench3.1/audio/fr/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
171
+ bench3.1/audio/fr/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
172
+ bench3.1/audio/fr/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
173
+ bench3.1/audio/fr/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
174
+ bench3.1/audio/fr/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
175
+ bench3.1/audio/fr/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
176
+ bench3.1/audio/fr/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
177
+ bench3.1/audio/it/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
178
+ bench3.1/audio/it/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
179
+ bench3.1/audio/it/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
180
+ bench3.1/audio/it/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
181
+ bench3.1/audio/it/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
182
+ bench3.1/audio/it/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
183
+ bench3.1/audio/it/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
184
+ bench3.1/audio/it/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
185
+ bench3.1/audio/it/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
186
+ bench3.1/audio/it/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
187
+ bench3.1/audio/it/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
188
+ bench3.1/audio/it/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
189
+ bench3.1/audio/it/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
190
+ bench3.1/audio/it/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
191
+ bench3.1/audio/it/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
192
+ bench3.1/audio/it/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
193
+ bench3.1/audio/it/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
194
+ bench3.1/audio/it/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
195
+ bench3.1/audio/it/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
196
+ bench3.1/audio/it/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
197
+ bench3.1/audio/it/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
198
+ bench3.1/audio/it/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
199
+ bench3.1/audio/it/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
200
+ bench3.1/audio/it/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
201
+ bench3.1/audio/it/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
202
+ bench3.1/audio/it/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
203
+ bench3.1/audio/it/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
204
+ bench3.1/audio/ja/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
205
+ bench3.1/audio/ja/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
206
+ bench3.1/audio/ja/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
207
+ bench3.1/audio/ja/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
208
+ bench3.1/audio/ja/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
209
+ bench3.1/audio/ja/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
210
+ bench3.1/audio/ja/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
211
+ bench3.1/audio/ja/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
212
+ bench3.1/audio/ja/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
213
+ bench3.1/audio/ja/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
214
+ bench3.1/audio/ja/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
215
+ bench3.1/audio/ja/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
216
+ bench3.1/audio/ja/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
217
+ bench3.1/audio/ja/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
218
+ bench3.1/audio/ja/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
219
+ bench3.1/audio/ja/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
220
+ bench3.1/audio/ja/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
221
+ bench3.1/audio/ja/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
222
+ bench3.1/audio/ja/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
223
+ bench3.1/audio/ja/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
224
+ bench3.1/audio/ja/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
225
+ bench3.1/audio/ja/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
226
+ bench3.1/audio/ja/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
227
+ bench3.1/audio/ja/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
228
+ bench3.1/audio/ja/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
229
+ bench3.1/audio/ja/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
230
+ bench3.1/audio/ja/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
231
+ bench3.1/audio/ko/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
232
+ bench3.1/audio/ko/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
233
+ bench3.1/audio/ko/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
234
+ bench3.1/audio/ko/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
235
+ bench3.1/audio/ko/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
236
+ bench3.1/audio/ko/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
237
+ bench3.1/audio/ko/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
238
+ bench3.1/audio/ko/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
239
+ bench3.1/audio/ko/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
240
+ bench3.1/audio/ko/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
241
+ bench3.1/audio/ko/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
242
+ bench3.1/audio/ko/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
243
+ bench3.1/audio/ko/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
244
+ bench3.1/audio/ko/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
245
+ bench3.1/audio/ko/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
246
+ bench3.1/audio/ko/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
247
+ bench3.1/audio/ko/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
248
+ bench3.1/audio/ko/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
249
+ bench3.1/audio/ko/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
250
+ bench3.1/audio/ko/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
251
+ bench3.1/audio/ko/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
252
+ bench3.1/audio/ko/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
253
+ bench3.1/audio/ko/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
254
+ bench3.1/audio/ko/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
255
+ bench3.1/audio/ko/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
256
+ bench3.1/audio/ko/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
257
+ bench3.1/audio/ko/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
258
+ bench3.1/audio/pt/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
259
+ bench3.1/audio/pt/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
260
+ bench3.1/audio/pt/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
261
+ bench3.1/audio/pt/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
262
+ bench3.1/audio/pt/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
263
+ bench3.1/audio/pt/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
264
+ bench3.1/audio/pt/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
265
+ bench3.1/audio/pt/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
266
+ bench3.1/audio/pt/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
267
+ bench3.1/audio/pt/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
268
+ bench3.1/audio/pt/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
269
+ bench3.1/audio/pt/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
270
+ bench3.1/audio/pt/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
271
+ bench3.1/audio/pt/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
272
+ bench3.1/audio/pt/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
273
+ bench3.1/audio/pt/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
274
+ bench3.1/audio/pt/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
275
+ bench3.1/audio/pt/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
276
+ bench3.1/audio/pt/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
277
+ bench3.1/audio/pt/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
278
+ bench3.1/audio/pt/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
279
+ bench3.1/audio/pt/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
280
+ bench3.1/audio/pt/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
281
+ bench3.1/audio/pt/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
282
+ bench3.1/audio/pt/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
283
+ bench3.1/audio/pt/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
284
+ bench3.1/audio/pt/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
285
+ bench3.1/audio/ru/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
286
+ bench3.1/audio/ru/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
287
+ bench3.1/audio/ru/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
288
+ bench3.1/audio/ru/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
289
+ bench3.1/audio/ru/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
290
+ bench3.1/audio/ru/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
291
+ bench3.1/audio/ru/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
292
+ bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
293
+ bench3.1/audio/ru/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
294
+ bench3.1/audio/ru/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
295
+ bench3.1/audio/ru/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
296
+ bench3.1/audio/ru/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
297
+ bench3.1/audio/ru/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
298
+ bench3.1/audio/ru/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
299
+ bench3.1/audio/ru/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
300
+ bench3.1/audio/ru/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
301
+ bench3.1/audio/ru/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
302
+ bench3.1/audio/ru/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
303
+ bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
304
+ bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
305
+ bench3.1/audio/ru/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
306
+ bench3.1/audio/ru/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
307
+ bench3.1/audio/ru/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
308
+ bench3.1/audio/ru/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
309
+ bench3.1/audio/ru/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
310
+ bench3.1/audio/ru/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
311
+ bench3.1/audio/ru/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
312
+ bench3.1/audio/zh/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
313
+ bench3.1/audio/zh/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
314
+ bench3.1/audio/zh/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
315
+ bench3.1/audio/zh/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
316
+ bench3.1/audio/zh/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
317
+ bench3.1/audio/zh/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
318
+ bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
319
+ bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
320
+ bench3.1/audio/zh/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
321
+ bench3.1/audio/zh/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
322
+ bench3.1/audio/zh/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
323
+ bench3.1/audio/zh/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
324
+ bench3.1/audio/zh/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
325
+ bench3.1/audio/zh/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
326
+ bench3.1/audio/zh/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
327
+ bench3.1/audio/zh/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
328
+ bench3.1/audio/zh/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
329
+ bench3.1/audio/zh/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
330
+ bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
331
+ bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
332
+ bench3.1/audio/zh/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
333
+ bench3.1/audio/zh/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
334
+ bench3.1/audio/zh/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
335
+ bench3.1/audio/zh/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
336
+ bench3.1/audio/zh/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
337
+ bench3.1/audio/zh/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
338
+ bench3.1/audio/zh/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
339
+ bench3.1/images/brutes/p01_32b.png filter=lfs diff=lfs merge=lfs -text
340
+ bench3.1/images/brutes/p01_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
341
+ bench3.1/images/brutes/p01_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
342
+ bench3.1/images/brutes/p01_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
343
+ bench3.1/images/brutes/p01_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
344
+ bench3.1/images/brutes/p02_32b.png filter=lfs diff=lfs merge=lfs -text
345
+ bench3.1/images/brutes/p02_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
346
+ bench3.1/images/brutes/p02_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
347
+ bench3.1/images/brutes/p02_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
348
+ bench3.1/images/brutes/p02_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
349
+ bench3.1/images/brutes/p03_32b.png filter=lfs diff=lfs merge=lfs -text
350
+ bench3.1/images/brutes/p03_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
351
+ bench3.1/images/brutes/p03_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
352
+ bench3.1/images/brutes/p03_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
353
+ bench3.1/images/brutes/p03_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
354
+ bench3.1/images/brutes/p04_32b.png filter=lfs diff=lfs merge=lfs -text
355
+ bench3.1/images/brutes/p04_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
356
+ bench3.1/images/brutes/p04_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
357
+ bench3.1/images/brutes/p04_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
358
+ bench3.1/images/brutes/p04_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
359
+ bench3.1/images/brutes/p05_32b.png filter=lfs diff=lfs merge=lfs -text
360
+ bench3.1/images/brutes/p05_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
361
+ bench3.1/images/brutes/p05_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
362
+ bench3.1/images/brutes/p05_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
363
+ bench3.1/images/brutes/p05_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
364
+ bench3.1/images/brutes/p06_32b.png filter=lfs diff=lfs merge=lfs -text
365
+ bench3.1/images/brutes/p06_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
366
+ bench3.1/images/brutes/p06_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
367
+ bench3.1/images/brutes/p06_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
368
+ bench3.1/images/brutes/p06_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
369
+ bench3.1/images/brutes/p07_32b.png filter=lfs diff=lfs merge=lfs -text
370
+ bench3.1/images/brutes/p07_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
371
+ bench3.1/images/brutes/p07_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
372
+ bench3.1/images/brutes/p07_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
373
+ bench3.1/images/brutes/p07_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
374
+ bench3.1/images/brutes/p08_32b.png filter=lfs diff=lfs merge=lfs -text
375
+ bench3.1/images/brutes/p08_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
376
+ bench3.1/images/brutes/p08_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
377
+ bench3.1/images/brutes/p08_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
378
+ bench3.1/images/brutes/p08_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
379
+ bench3.1/images/brutes/p09_32b.png filter=lfs diff=lfs merge=lfs -text
380
+ bench3.1/images/brutes/p09_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
381
+ bench3.1/images/brutes/p09_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
382
+ bench3.1/images/brutes/p09_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
383
+ bench3.1/images/brutes/p09_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
384
+ bench3.1/images/brutes/p10_32b.png filter=lfs diff=lfs merge=lfs -text
385
+ bench3.1/images/brutes/p10_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
386
+ bench3.1/images/brutes/p10_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
387
+ bench3.1/images/brutes/p10_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
388
+ bench3.1/images/brutes/p10_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
389
+ bench3.1/images/brutes/p11_32b.png filter=lfs diff=lfs merge=lfs -text
390
+ bench3.1/images/brutes/p11_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
391
+ bench3.1/images/brutes/p11_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
392
+ bench3.1/images/brutes/p11_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
393
+ bench3.1/images/brutes/p11_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
394
+ bench3.1/images/brutes/p12_32b.png filter=lfs diff=lfs merge=lfs -text
395
+ bench3.1/images/brutes/p12_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
396
+ bench3.1/images/brutes/p12_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
397
+ bench3.1/images/brutes/p12_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
398
+ bench3.1/images/brutes/p12_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
399
+ bench3.1/images/brutes/p13_32b.png filter=lfs diff=lfs merge=lfs -text
400
+ bench3.1/images/brutes/p13_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
401
+ bench3.1/images/brutes/p13_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
402
+ bench3.1/images/brutes/p13_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
403
+ bench3.1/images/brutes/p13_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
404
+ bench3.1/images/brutes/p14_32b.png filter=lfs diff=lfs merge=lfs -text
405
+ bench3.1/images/brutes/p14_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
406
+ bench3.1/images/brutes/p14_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
407
+ bench3.1/images/brutes/p14_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
408
+ bench3.1/images/brutes/p14_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
409
+ bench3.1/images/brutes/p15_32b.png filter=lfs diff=lfs merge=lfs -text
410
+ bench3.1/images/brutes/p15_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
411
+ bench3.1/images/brutes/p15_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
412
+ bench3.1/images/brutes/p15_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
413
+ bench3.1/images/brutes/p15_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
414
+ bench3.1/images/planches/p01.jpg filter=lfs diff=lfs merge=lfs -text
415
+ bench3.1/images/planches/p02.jpg filter=lfs diff=lfs merge=lfs -text
416
+ bench3.1/images/planches/p03.jpg filter=lfs diff=lfs merge=lfs -text
417
+ bench3.1/images/planches/p04.jpg filter=lfs diff=lfs merge=lfs -text
418
+ bench3.1/images/planches/p05.jpg filter=lfs diff=lfs merge=lfs -text
419
+ bench3.1/images/planches/p06.jpg filter=lfs diff=lfs merge=lfs -text
420
+ bench3.1/images/planches/p07.jpg filter=lfs diff=lfs merge=lfs -text
421
+ bench3.1/images/planches/p08.jpg filter=lfs diff=lfs merge=lfs -text
422
+ bench3.1/images/planches/p09.jpg filter=lfs diff=lfs merge=lfs -text
423
+ bench3.1/images/planches/p10.jpg filter=lfs diff=lfs merge=lfs -text
424
+ bench3.1/images/planches/p12.jpg filter=lfs diff=lfs merge=lfs -text
425
+ bench3.1/images/planches/p13.jpg filter=lfs diff=lfs merge=lfs -text
426
+ bench3.1/images/planches/p14.jpg filter=lfs diff=lfs merge=lfs -text
427
+ bench3.1/images/planches/p15.jpg filter=lfs diff=lfs merge=lfs -text
428
+ bench3.1/video/clipproj-v3.1-eleven-languages.mp4 filter=lfs diff=lfs merge=lfs -text
429
+ bench3.1/video/langues/ar/32b.mp4 filter=lfs diff=lfs merge=lfs -text
430
+ bench3.1/video/langues/ar/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
431
+ bench3.1/video/langues/ar/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
432
+ bench3.1/video/langues/ar/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
433
+ bench3.1/video/langues/ar/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
434
+ bench3.1/video/langues/ar/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
435
+ bench3.1/video/langues/ar/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
436
+ bench3.1/video/langues/ar/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
437
+ bench3.1/video/langues/ar/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
438
+ bench3.1/video/langues/de/32b.mp4 filter=lfs diff=lfs merge=lfs -text
439
+ bench3.1/video/langues/de/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
440
+ bench3.1/video/langues/de/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
441
+ bench3.1/video/langues/de/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
442
+ bench3.1/video/langues/de/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
443
+ bench3.1/video/langues/de/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
444
+ bench3.1/video/langues/de/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
445
+ bench3.1/video/langues/de/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
446
+ bench3.1/video/langues/de/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
447
+ bench3.1/video/langues/en/32b.mp4 filter=lfs diff=lfs merge=lfs -text
448
+ bench3.1/video/langues/en/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
449
+ bench3.1/video/langues/en/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
450
+ bench3.1/video/langues/en/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
451
+ bench3.1/video/langues/en/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
452
+ bench3.1/video/langues/en/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
453
+ bench3.1/video/langues/en/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
454
+ bench3.1/video/langues/en/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
455
+ bench3.1/video/langues/en/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
456
+ bench3.1/video/langues/es/32b.mp4 filter=lfs diff=lfs merge=lfs -text
457
+ bench3.1/video/langues/es/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
458
+ bench3.1/video/langues/es/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
459
+ bench3.1/video/langues/es/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
460
+ bench3.1/video/langues/es/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
461
+ bench3.1/video/langues/es/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
462
+ bench3.1/video/langues/es/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
463
+ bench3.1/video/langues/es/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
464
+ bench3.1/video/langues/es/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
465
+ bench3.1/video/langues/fr/32b.mp4 filter=lfs diff=lfs merge=lfs -text
466
+ bench3.1/video/langues/fr/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
467
+ bench3.1/video/langues/fr/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
468
+ bench3.1/video/langues/fr/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
469
+ bench3.1/video/langues/fr/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
470
+ bench3.1/video/langues/fr/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
471
+ bench3.1/video/langues/fr/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
472
+ bench3.1/video/langues/fr/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
473
+ bench3.1/video/langues/fr/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
474
+ bench3.1/video/langues/it/32b.mp4 filter=lfs diff=lfs merge=lfs -text
475
+ bench3.1/video/langues/it/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
476
+ bench3.1/video/langues/it/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
477
+ bench3.1/video/langues/it/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
478
+ bench3.1/video/langues/it/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
479
+ bench3.1/video/langues/it/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
480
+ bench3.1/video/langues/it/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
481
+ bench3.1/video/langues/it/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
482
+ bench3.1/video/langues/it/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
483
+ bench3.1/video/langues/ja/32b.mp4 filter=lfs diff=lfs merge=lfs -text
484
+ bench3.1/video/langues/ja/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
485
+ bench3.1/video/langues/ja/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
486
+ bench3.1/video/langues/ja/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
487
+ bench3.1/video/langues/ja/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
488
+ bench3.1/video/langues/ja/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
489
+ bench3.1/video/langues/ja/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
490
+ bench3.1/video/langues/ja/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
491
+ bench3.1/video/langues/ja/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
492
+ bench3.1/video/langues/ko/32b.mp4 filter=lfs diff=lfs merge=lfs -text
493
+ bench3.1/video/langues/ko/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
494
+ bench3.1/video/langues/ko/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
495
+ bench3.1/video/langues/ko/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
496
+ bench3.1/video/langues/ko/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
497
+ bench3.1/video/langues/ko/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
498
+ bench3.1/video/langues/ko/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
499
+ bench3.1/video/langues/ko/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
500
+ bench3.1/video/langues/ko/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
501
+ bench3.1/video/langues/pt/32b.mp4 filter=lfs diff=lfs merge=lfs -text
502
+ bench3.1/video/langues/pt/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
503
+ bench3.1/video/langues/pt/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
504
+ bench3.1/video/langues/pt/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
505
+ bench3.1/video/langues/pt/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
506
+ bench3.1/video/langues/pt/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
507
+ bench3.1/video/langues/pt/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
508
+ bench3.1/video/langues/pt/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
509
+ bench3.1/video/langues/pt/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
510
+ bench3.1/video/langues/ru/32b.mp4 filter=lfs diff=lfs merge=lfs -text
511
+ bench3.1/video/langues/ru/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
512
+ bench3.1/video/langues/ru/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
513
+ bench3.1/video/langues/ru/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
514
+ bench3.1/video/langues/ru/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
515
+ bench3.1/video/langues/ru/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
516
+ bench3.1/video/langues/ru/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
517
+ bench3.1/video/langues/ru/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
518
+ bench3.1/video/langues/ru/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
519
+ bench3.1/video/langues/zh/32b.mp4 filter=lfs diff=lfs merge=lfs -text
520
+ bench3.1/video/langues/zh/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
521
+ bench3.1/video/langues/zh/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
522
+ bench3.1/video/langues/zh/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
523
+ bench3.1/video/langues/zh/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
524
+ bench3.1/video/langues/zh/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
525
+ bench3.1/video/langues/zh/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
526
+ bench3.1/video/langues/zh/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
527
+ bench3.1/video/langues/zh/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,234 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ tags:
4
+ - comfyui
5
+ - minimax-h3
6
+ - text-to-video
7
+ - qwen3-vl
8
+ - text-encoder
9
+ base_model:
10
+ - Comfy-Org/MiniMax-H3
11
+ - Qwen/Qwen3-VL-4B-Instruct
12
+ library_name: comfyui
13
+ ---
14
+
15
+ # ClipProj — MiniMax H3 conditioning from a Qwen3-VL-4B or 8B
16
+
17
+ **Projection matrices that let a small Qwen3-VL replace the Qwen3-VL-32B text encoder of MiniMax H3.**
18
+
19
+ **15.7 GB → 4.5 GB of VRAM**, with no change to the diffusion model, the VAEs or the sampler.
20
+
21
+ ---
22
+
23
+ ## v3.1 — better multilingual speech
24
+
25
+ **No measurable gain in image over v3 — the whole gain is in speech.** The image scores of the two generations overlap entirely, and nothing measured here separates either of them from the 32B, which is not the same as saying they are identical to it: the cosine is blind to countable attributes. What v3.1 improves is pronunciation in the ten languages that are not English — **29 %% fewer phoneme errors overall, 60 to 74 %% fewer in Spanish, French, German and Italian.**
26
+
27
+ <video controls width="360" src="https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4"></video>
28
+
29
+ *Eleven languages, 88 seconds. For each one the **smallest** file that matches the 32B — not the best one. Nine of the eleven run on a 4B. [Direct link](https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4)*
30
+
31
+ > ⚠️ **The video looks and sounds rough, on purpose.** 0.3 MP, 6 sampling steps — the settings that made 297 renders affordable — then upscaled. The audio is tinny for the same reason, **identically so on the 32B**, with the same 19 dB dip between 1 and 3 kHz. This is a pronunciation test, not a showcase: it exists to let you hear *which words come out*.
32
+
33
+ **Four new files:** `mmh3-4b-ClipProj-v3.1`, `mmh3-4b-ClipProj-v3.1-mlp`, `mmh3-8b-ClipProj-v3.1`, `mmh3-8b-ClipProj-v3.1-mlp`. They load on node **0.1.13** with no code change — same base as v3. Start with **`mmh3-4b-ClipProj-v3.1`**: 26 MB of projection, 4.6 GB with its encoder.
34
+
35
+ **What changed:** the calibration corpus now gives every writing system a comparable share instead of being overwhelmingly English, with raw Arabic text added. Phoneme errors drop **29 % overall**, 60 to 74 % on the European languages.
36
+
37
+ **What it is measured against:** the 32B does not reproduce itself. Change nothing but the seed and it re-pronounces **5.8 phonemes out of 75** differently. The four v3.1 files sit at 6.4 to 7.0 — so swapping the encoder costs about what re-rolling the seed costs. Every figure is normalised against that variance rather than against zero, and on that scale nothing separates 4B from 8B, or ridge from residual.
38
+
39
+ **Full report** — three seeds, 297 speech renders, 405 image renders, method, limitations and all the raw data: **[bench3.1/README.md](./bench3.1/README.md)**
40
+
41
+ ---
42
+
43
+
44
+ > ⚠️ **Proof of concept — working, but a proof of concept.** It runs and produces good video, and every number below was measured on real hardware. Built and tested on a single setup (Windows 11, NVIDIA, ComfyUI 0.31.0) with deliberately limited exploration.
45
+
46
+ > **Arriving from a tutorial or an article?** Anything published before 11 August names files that have moved. `h3_qwen3vl_4b_tap24`, `h3_control_zero` and `h3_control_identity` are still here, one folder down in `obsolete/`, so nothing is lost. But take the current set instead: **`mmh3-4b-ClipProj-celeb-mlp`** for a Qwen3-VL-4B, **`mmh3-8b-ClipProj-celeb-mlp`** for an 8B. They are better on every measurement below, and they need node **0.1.4 or later**.
47
+
48
+ These files are useless on their own. They require the custom node:
49
+ **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
50
+
51
+ ## Where this came from
52
+
53
+ I am not an ML researcher. I work in imaging, and programming is a tool and a hobby rather than my trade. This started as something to tinker with: I wanted to understand how a diffusion model actually uses its text encoder, and the only way I know how to understand something is to take it apart and see whether it still runs afterwards.
54
+
55
+ So the question was never "how do I save VRAM". It was "is this even possible at all". I expected it to fail. A linear map between two models that were never trained together, fitted in a single pass with no gradients and no learning rate, has no business producing usable video.
56
+
57
+ It did, and the first results were good enough that keeping them on my own disk seemed silly. That is the whole story, and it is why this is labelled a proof of concept rather than a tool: it was never designed as one.
58
+
59
+ It is also why there are so many measurements on the model card. Before showing this to anyone I had to convince myself I was not fooling myself, and most of what I tried along the way turned out to be wrong. Those attempts are written down as well, in [MEASUREMENTS.md](https://github.com/nicolab28/ComfyUI-ClipProj/blob/main/MEASUREMENTS.md) and [CALIBRATION.md](https://github.com/nicolab28/ComfyUI-ClipProj/blob/main/CALIBRATION.md).
60
+
61
+ ## Update to node 0.1.4, and re-download the `-mlp` matrices
62
+
63
+ **The `-mlp` matrices are now fp16 and half the size.** The residual network was published in fp32 and the node forced fp32 on load regardless of the file, so storing it in half precision would have halved the download and saved nothing at all in VRAM. Node 0.1.4 keeps a residual in whatever precision it was saved in, converting its inputs and outputs around it instead. Measured: 240 MB on the card instead of 480 for the 4B, 288 instead of 576 for the 8B. The files here have been replaced under the same names — re-download them, and take 0.1.4 with them, because an older node will load them and cast them straight back up to fp32.
64
+
65
+ Node 0.1.4 also frees the card **before** loading a replacement encoder rather than after, which matters if yours is tight enough that two encoders will not sit on it at once.
66
+
67
+ ## Also in 0.1.3
68
+
69
+ Two reasons, one of them silent.
70
+
71
+ **The `-mlp` matrices carry a residual network, and an older node ignores it without saying so.** It reads the matrix, finds keys it does not know, drops them, and applies the linear part alone. Nothing fails, nothing warns, and you end up judging the plain matrix while believing you tested the residual. Node 0.1.3 reads them.
72
+
73
+ **Everything is renamed.** The old `h3_qwen3vl_*` files have moved to `obsolete/` and the `.pt` copies are gone: opening a pickle executes code, which makes no sense for a file holding six tensors. If a workflow of yours names an old file, either point it at `obsolete/` or, better, switch to the new set.
74
+
75
+ ## What this is
76
+
77
+ MiniMax H3 conditions on a Qwen3-VL-32B truncated to 50 layers — 15.7 GB in NVFP4 — solely to turn a prompt into a `[seq, 5120]` tensor. This repository provides a learned map that lets a much smaller Qwen3-VL produce the same conditioning:
78
+
79
+ ```
80
+ cond = ((h - mean_in) / std_in) @ W * std_out + mean_out
81
+ ```
82
+
83
+ and, in the `-mlp` files, plus the output of a small residual network fed the same standardised input.
84
+
85
+ It works because every Qwen3-VL shares the **same tokenizer** (151936 tokens): a prompt yields the same tokens at the same positions in both models, so a position-by-position mapping between their hidden states can be learned. The matrix is fitted by plain **ridge regression** — no gradients, no epochs, no learning rate. The residual network is the only part that is trained.
86
+
87
+ ## Files
88
+
89
+ Put them in `ComfyUI/models/clip_projections/`.
90
+
91
+ **Start with `mmh3-8b-ClipProj-celeb-mlp` if you have the VRAM, `mmh3-4b-ClipProj-celeb-mlp` otherwise.**
92
+
93
+ | File | Encoder | Names covered | Residual | Test cosine |
94
+ |---|---|---|---|---|
95
+ | `mmh3-4b-ClipProj` | any Qwen3-VL-4B | no | no | 0.7169 |
96
+ | `mmh3-4b-ClipProj-mlp` | any Qwen3-VL-4B | no | yes | 0.7944 |
97
+ | `mmh3-4b-ClipProj-celeb` | any Qwen3-VL-4B | **yes** | no | 0.7095 |
98
+ | `mmh3-4b-ClipProj-celeb-mlp` | any Qwen3-VL-4B | **yes** | yes | 0.7930 |
99
+ | `mmh3-8b-ClipProj` | any Qwen3-VL-8B | no | no | 0.7528 |
100
+ | `mmh3-8b-ClipProj-mlp` | any Qwen3-VL-8B | no | yes | 0.7970 |
101
+ | `mmh3-8b-ClipProj-celeb` | any Qwen3-VL-8B | **yes** | no | 0.7466 |
102
+ | `mmh3-8b-ClipProj-celeb-mlp` | any Qwen3-VL-8B | **yes** | yes | **0.8037** |
103
+ | `mmh3-ClipProj-control-zero` | — | control, run it once | — | — |
104
+ | `mmh3-ClipProj-control-identity` | — | control, run it once | — | — |
105
+
106
+ All eight are calibrated on the same general corpus and measured on the same held-out prompts, so the column is comparable across every row.
107
+
108
+ Every matrix works on **any variant of its own size**: the measured cosine gap between a bf16-calibrated matrix applied to an abliterated fp8 encoder is 0.0023. You do not need the exact checkpoint a matrix was calibrated on. The 8B matrices need an 8B encoder though — 4096 input dimensions instead of 2560 — and the node checks the width and refuses a mismatch.
109
+
110
+ ## Named people
111
+
112
+ **This is what changed in 0.1.3, and it was a corpus problem.**
113
+
114
+ The calibration corpus named a person on about 70 lines out of 8632, roughly 0.02 % of the training tokens. The directions of the hidden space that carry an identity were therefore constrained by nothing at all, and the fit put whatever minimised the error on landscape descriptions there. Named people came out as somebody else.
115
+
116
+ The `-celeb` matrices add 500 people, ranked by popularity, with five short prompts and two long ones each. What it buys and what it costs:
117
+
118
+ | | name tokens | rest of the sentence | general test set |
119
+ |---|---|---|---|
120
+ | without | 0.8265 | 0.9358 | 0.7944 |
121
+ | with | **0.8844** | **0.9516** | 0.7930 |
122
+
123
+ Seven thousandths of cosine on the general corpus, for six points on the tokens that carry an identity. The rest of the sentence improves too, because the celebrity prompts are short and the general corpus had nothing under fifteen words.
124
+
125
+ Two findings that decide how far this is worth pushing.
126
+
127
+ **Two contexts per person are enough.** Measured on contexts held out for people the matrix had seen: 0.9875 at two, 0.9945 at five, 0.9986 at twenty. Forty is a waste.
128
+
129
+ **Five hundred names generalise to names never seen.** A held-out band at popularity ranks 501 to 540, absent from every calibration, reconstructs at 0.8795 against 0.8844 for the covered ones. Covering 500 people does not teach 500 names; it teaches the map how to handle that region of the space. Going to several thousand would buy very little.
130
+
131
+ **What still fails is not the corpus.** Characters whose identity is a mask rather than a face come out as a stranger wearing the right costume. People whose fame predates the era when everything was photographed come out wrong or generic. And some names fail on the plain 32B too, so run the reference before blaming the projection — that check has overturned three of my own conclusions.
132
+
133
+ ## Where the calibration data comes from
134
+
135
+ The general corpus is [GokuScraper/seedance-2-prompts-datasets](https://huggingface.co/datasets/GokuScraper/seedance-2-prompts-datasets), filtered to prompts of fifteen words or more and deduplicated: 8632 lines, median 128 words. The 500 named people come from a TMDB export published on Kaggle, ranked by popularity, with transliterated names dropped beyond rank 1000.
136
+
137
+ Around each name, five short prompts are generated from templates, and two longer ones in MiniMax H3's section format are written by Mistral Small and Gemini Flash Lite, half each. Everything needed to rebuild the corpus is in the node's `calibration/` folder, including the system prompt the long prompts were written from.
138
+
139
+ ## The residual network
140
+
141
+ The `-mlp` files carry a `d_in → 16384 → 5120` network with a GELU, added to the matrix rather than replacing it. Its last layer is initialised to zero, so at the first step the model reproduces the matrix exactly and can only improve on it. It is worth 0.05 to 0.08 of cosine, four times what multiplying the corpus by eleven buys the linear map.
142
+
143
+ **Which of the two renders better is not settled.** The cosine does not predict it — that is the single most repeated lesson of this project. Try both on your own prompts.
144
+
145
+ Two things measured while building it. Width beats depth: at equal parameter count, two hidden layers of 8192 reach 0.7691 against 0.7944 for one layer of 16384. And a residual extrapolates worse than a matrix does — outside the corpus it saw, a linear map degrades gracefully while the network collapses.
146
+
147
+ ## Measured results
148
+
149
+ | | 4B | 8B |
150
+ |---|---|---|
151
+ | matrix, no names | 0.7169 | 0.7528 |
152
+ | matrix + residual | 0.7944 | 0.7970 |
153
+ | matrix, names covered | 0.7095 | 0.7466 |
154
+ | matrix + residual, names covered | 0.7930 | **0.8037** |
155
+
156
+ A cosine of 0.79 sounds poor and is not — the DiT tolerates far more than the metric suggests. What holds up in actual generation: simple prompts, structured multi-shot prompts with several distinct cuts and no bleed between them, fl2va with first and last frame, ref2va with a reference image, and since 0.1.3 ref2va with a reference video.
157
+
158
+ Fidelity does **not** collapse on short prompts: measured per-token cosine goes from 0.937 at 80 words to 0.908 at 2 words, once the attention sink is handled.
159
+
160
+ ## Speech
161
+
162
+ The first release lost non-English speech: a French line came out half Spanish, and the 8B put everything in English. That was the clearest regression and I could not explain it then.
163
+
164
+ With `mmh3-8b-ClipProj-celeb-mlp`, a three-shot clip carrying English, French and Spanish comes out like the 32B does, and the audio level gap measured against the reference has gone from 7.6 dB to 3.5.
165
+
166
+ Part of what was blamed on the projection was not the projection. A line that fills more than about two thirds of its shot comes out slurred whatever encoder produced the conditioning — the fix is a longer shot, not a better matrix. And MiniMax H3 expects speech wrapped in `<d>[Language] ...</d>` with a stable speaker id declared beforehand; without that, one voice with one accent is used for the whole clip. Neither of those is documented here because neither is ours, but both cost me a day.
167
+
168
+ ## Run the controls first
169
+
170
+ The two control matrices exist to prove the learned matrix is doing the work rather than the diffusion model. Same prompt, same seed, only the matrix changes:
171
+
172
+ | Matrix | Output for *"a red ball on a wood table"* |
173
+ |---|---|
174
+ | `mmh3-ClipProj-control-zero` | a countryside landscape — the prompt is entirely ignored |
175
+ | `mmh3-ClipProj-control-identity` | a golden object in flames — unusable |
176
+ | a learned matrix | the red ball on a wood table |
177
+
178
+ `‖W_identity‖ = 50.6` against `‖W_learned‖ = 52.4` — near-identical energy, so the difference is structural, not a matter of scale.
179
+
180
+ **If the identity control ever looks fine, the learned matrix adds nothing — and you want to know that before trusting it.**
181
+
182
+ ## What is in obsolete/
183
+
184
+ The previous matrices, kept because a comparison posted on r/StableDiffusion ran on them and the links have to keep working. They have no name coverage and are calibrated on a corpus thirty times smaller. There is no reason to prefer them.
185
+
186
+ Among them, the `CONDPROJ` pair, and the story is worth telling because the mistake was instructive.
187
+
188
+ The DiT does not consume the conditioning as it arrives: it first passes it through `condition_proj`, a `Linear(5120 → 5376)` feeding the token refiner. That layer's spectrum is very uneven — a factor of 45 between the top and bottom deciles of its singular values, 52 % of the energy in 10 % of the directions. Plain ridge regression ignores this and spends as much effort on a direction the DiT will multiply by 0.10 as on one it will multiply by 37. Calibrating against the **output** of that layer instead, then mapping back through the pseudo-inverse, should therefore minimise the error the DiT actually sees. The cosine went from 0.697 to 0.845 on the 4B and 0.731 to 0.860 on the 8B.
189
+
190
+ Then I compared what the two matrices actually output:
191
+
192
+ ```
193
+ 4B CONDPROJ against unweighted, same corpus cosine 0.999998
194
+ 8B CONDPROJ against unweighted, same corpus cosine 0.999999
195
+ ```
196
+
197
+ They are the same function. Unregularised least squares is invariant to an invertible linear transform of the targets, so fitting in one space and mapping back recovers the same map; only the ridge penalty breaks that invariance, and with 37 851 training tokens against λ = 1000 it barely binds. The entire gain was an artefact of measuring in a different space.
198
+
199
+ *The idea came from u/stddealer on r/StableDiffusion, and it was a good one. The measurement is on me: I published the cosine before checking whether the matrix had changed at all.*
200
+
201
+ ## Known limitations
202
+
203
+ **Quantisation costs facts.** Comparing `int8_convrot` against `bf16` on factual recall shows errors appearing under quantisation. Fine for general use, worth knowing if your prompts lean on proper nouns.
204
+
205
+ **Masks defeat identity.** A character recognised by a costume rather than a face comes out as an unknown person in the right suit. No corpus fixes that, because the identity is not in the name's representation to begin with.
206
+
207
+ **Counting is unreliable, and not because of the projection.** Ask for three of something and you get four, on the 32B too. Enumerating works better than announcing a number.
208
+
209
+ ## Required models
210
+
211
+ | Role | Model |
212
+ |---|---|
213
+ | Diffusion model + VAEs | [Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3) |
214
+ | Text encoder, 4B | [Comfy-Org/Krea-2](https://huggingface.co/Comfy-Org/Krea-2) → `text_encoders/qwen3vl_4b_fp8_scaled.safetensors` |
215
+ | Text encoder, 8B | any ComfyUI-format Qwen3-VL-8B (the 8B matrices expect 4096 input dims) |
216
+
217
+ The 32B text encoder is **no longer needed** — that is the entire point.
218
+
219
+ ## Licence and responsibility
220
+
221
+ These matrices are released under **MIT**, like the node.
222
+
223
+ They are derived from the activations of both models, and their legal status is unclear. They are provided as-is, for research, with no claim of ownership over anything derived from the underlying models.
224
+
225
+ - **Qwen3-VL** is published by Alibaba under **Apache 2.0**. Read and comply with its terms and acceptable-use policy.
226
+ - **MiniMax H3** ships under a **custom licence**. Read it before any use, particularly commercial.
227
+
228
+ This project is **not affiliated with, endorsed by, or connected to** Alibaba / Qwen, MiniMax, or Comfy Org.
229
+
230
+ You remain responsible for what you generate and for complying with the licences of every model you load.
231
+
232
+ ## Credits
233
+
234
+ Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number quoted was measured on real hardware, not estimated: where a prediction turned out wrong, the measurement won and the text was corrected. Three claims in the previous version of this file were wrong and are corrected here.
RELEASE_NOTES_v3.md ADDED
@@ -0,0 +1,160 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # ClipProj v3
2
+
3
+ Four new matrices, calibrated against `qwen3vl_32b_minimax_h3_nvfp4_awq` — the
4
+ base MiniMax H3 encoder, the one a plain `Load CLIP` gives you.
5
+
6
+ Requires node **0.1.13** or later. The `-v3-mlp` files carry no linear matrix,
7
+ and an older node raises `KeyError: 'W'` on load.
8
+
9
+ ## What changed
10
+
11
+ **Body descriptions.** While testing v2 we found that naming one part of a
12
+ body could rewrite the whole of it — build, height and face shifting together,
13
+ none of it asked for. Those matrices were calibrated against a modified 32B.
14
+ v3 is calibrated against the stock encoder, and we no longer observe the
15
+ problem: a build and an attribute stated in the same prompt are honoured
16
+ independently.
17
+
18
+ **Everything is measurably closer to the 32B.** Measured on a single reference
19
+ prompt, every projection encoded against the same stock 32B, so the numbers are
20
+ comparable to each other — which the figures published with v1 and v2 were not,
21
+ having been measured against different targets and in different spaces. The
22
+ `cos_test` written inside each file is the training-time figure against that
23
+ campaign's own target and does **not** compare across versions; the comparable
24
+ number is stored separately as `cos_prompt_reference`.
25
+
26
+ The render pipeline was verified deterministic before measuring anything: the
27
+ same workflow run twice, thirteen days apart, produced two files with identical
28
+ SHA-256. Any difference below therefore comes from the projection and nothing
29
+ else.
30
+
31
+ | projection | mean cosine vs 32B |
32
+ |---|---|
33
+ | 8B v3-mlp | 0.9449 |
34
+ | 8B v2 `celeb-mlp` | 0.9393 |
35
+ | 4B v3-mlp | 0.9381 |
36
+ | 4B v2 `celeb-mlp` | 0.9293 |
37
+ | 8B v3 | 0.9289 |
38
+ | 4B v3 | 0.9193 |
39
+
40
+ **The 8B now sees image tokens.** Until v3 the image corpus had only ever been
41
+ encoded with a 4B student, so both 8B matrices projected vision tokens without
42
+ having seen a single one — while the node accepts a reference image. Measured on
43
+ 100 held-out images, on the raw conditioning the diffusion model actually
44
+ receives:
45
+
46
+ | | vision tokens | text in the same sequences |
47
+ |---|---|---|
48
+ | 8B residual, before | 0.7692 | 0.9085 |
49
+ | 8B residual, after | 0.8578 | 0.9605 |
50
+ | 8B matrix, before | 0.7845 | 0.8926 |
51
+ | 8B matrix, after | 0.8457 | 0.9361 |
52
+
53
+ That costs 0.0027 of pure-text cosine on the residual and 0.0013 on the matrix —
54
+ which is why the 8B figures above are slightly below what a text-only corpus
55
+ would have given.
56
+
57
+ **Prefer the `-mlp` files on the measurement, not on this scene.** They sit
58
+ closer to the 32B, 0.9449 against 0.9289 on the 8B. But on this prompt the five
59
+ renders are faithful, plain matrices included: the pose, the dress, the white
60
+ pieces, the cat, the straw hat, the laundry, the bouncing knee all hold on all
61
+ five. A tightly written prompt survives even the linear baseline, and the
62
+ difference between the files shows up in the numbers well before it shows up on
63
+ screen.
64
+
65
+ ## What this does not fix, and will not
66
+
67
+ A projection cannot invent information the small encoder never had. Five hours
68
+ of training on a 3090, two more to encode the dataset, six and a half million
69
+ tokens — none of that changes what a 4B model wrote down in the first place.
70
+ The result is an approximation of the 32B's conditioning, not a copy of it.
71
+
72
+ In practice the line falls here: **what the prompt states, the projection
73
+ carries; what the prompt leaves open, the model fills from its own prior.**
74
+ Constrain a scene tightly and the projected renders track the 32B closely.
75
+ Leave the set dressing unstated — a cat somewhere, laundry on a line, furniture
76
+ — and it will be furnished differently. That is not a defect to be tuned away,
77
+ it is what a 20-degree angle between two conditioning vectors looks like on
78
+ screen.
79
+
80
+ ## Training data
81
+
82
+ The aim was to activate as much of the encoder's weight space as possible
83
+ rather than to cover one domain deeply. A matrix only learns to project the
84
+ directions it has actually seen used, so the corpus deliberately mixes
85
+ registers, languages and lengths.
86
+
87
+ | source | tokens |
88
+ |---|---|
89
+ | cinematic video prompts | 1 342 987 |
90
+ | native H3 format, 4 length draws | 3 169 879 |
91
+ | explicit register | 544 073 |
92
+ | Chinese | 532 302 |
93
+ | celebrity prompts, long form | 314 516 |
94
+ | filler sequences | 149 917 |
95
+ | celebrity prompts, short form | 99 668 |
96
+ | images, 3 blocks, 1 700 images | 349 244 |
97
+ | **total, both students** | **6 502 586** |
98
+
99
+ Both students now see the same corpus, images included — which makes the 4B and
100
+ the 8B comparable to each other for the first time.
101
+
102
+ 3 331 prompts for fitting, one in fifty held out for measurement. Tap 24 on
103
+ both students. Sequence lengths are drawn at random rather than truncated to a
104
+ fixed size, so a given word appears at many different positions instead of
105
+ always the same ones.
106
+
107
+ ## The demo folder
108
+
109
+ `demo/` holds five renders of the same scene. Same prompt, same seed 42, same
110
+ 8 steps, same turbo LoRA, same DiT, same VAE — only the projection changes. The
111
+ pipeline is reproducible bit for bit, verified by running it twice and comparing
112
+ the decoded video and audio streams, so every difference between these five
113
+ comes from the projection and nothing else.
114
+
115
+ | file | conditioning |
116
+ |---|---|
117
+ | `chess-32b-reference.mp4` | Qwen3-VL-32B, 15.7 GB |
118
+ | `chess-8b-mlp.mp4` | Qwen3-VL-8B + `v3-mlp`, 10.0 GB |
119
+ | `chess-4b-mlp.mp4` | Qwen3-VL-4B + `v3-mlp`, 4.8 GB |
120
+ | `chess-8b-ridge.mp4` | Qwen3-VL-8B + `v3`, matrix only |
121
+ | `chess-4b-ridge.mp4` | Qwen3-VL-4B + `v3`, matrix only |
122
+ | `chess-comparison.mp4` | the five in sequence, labelled |
123
+ | `chess-prompt.txt` | the prompt, verbatim |
124
+
125
+ The comparison runs matrix-only first, then the 32B, then the residuals — so the
126
+ reference sits in the middle and each half is read against it.
127
+
128
+ **You will not reproduce these files byte for byte, and that is normal.** Noticed
129
+ while testing something else, so take it as an observation rather than a study:
130
+ the result depends on the model of GPU the encoder runs on. Four cards, one
131
+ prompt, one seed, four different outputs — while two different RTX 3090s gave
132
+ byte-identical video. Not the architecture either: the 3060 and the 3090 are
133
+ both Ampere and disagree. Encoding the same prompt on two cards gives
134
+ conditioning that matches to a relative error of 7 × 10⁻⁷; eight denoising steps
135
+ turn that into a different piece of furniture. On one machine everything here is
136
+ reproducible to the bit, which is what makes the five-way comparison meaningful.
137
+
138
+ Watch her knee. The prompt asks for it three times, ending on a sentence of its
139
+ own: *"Her knee never stops bouncing."* It is the most redundant instruction in
140
+ the text, it is a continuous involuntary motion with no narrative purpose, and
141
+ it is the clearest single indicator that a projection carried what was written.
142
+ Then watch the cat, the laundry and the furniture, which are named once and
143
+ anchored nowhere — those move, and they are supposed to.
144
+
145
+ One thing none of the five gets right, including the 32B: she lifts a knight and
146
+ does not put it back on the same square. Object permanence through an occluding
147
+ hand on a grid of sixty-four identical squares is a limit of the video model, not
148
+ of the conditioning. It is listed here so nobody attributes it to the projection.
149
+
150
+ ## Files
151
+
152
+ | file | size | structure | encoder |
153
+ |---|---|---|---|
154
+ | `mmh3-4b-ClipProj-v3-mlp.safetensors` | 503 MB | residual only | Qwen3-VL-4B |
155
+ | `mmh3-8b-ClipProj-v3-mlp.safetensors` | 604 MB | residual only | Qwen3-VL-8B |
156
+ | `mmh3-4b-ClipProj-v3.safetensors` | 26 MB | matrix only | Qwen3-VL-4B |
157
+ | `mmh3-8b-ClipProj-v3.safetensors` | 42 MB | matrix only | Qwen3-VL-8B |
158
+
159
+ The residuals use a hidden width of 32 768 against 16 384 in v2, which is where
160
+ the extra download size comes from.
bench3.1/README.md ADDED
@@ -0,0 +1,506 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ tags:
4
+ - comfyui
5
+ - minimax-h3
6
+ - text-to-video
7
+ - qwen3-vl
8
+ - text-encoder
9
+ - multilingual
10
+ base_model:
11
+ - Comfy-Org/MiniMax-H3
12
+ - Qwen/Qwen3-VL-4B-Instruct
13
+ - Qwen/Qwen3-VL-8B-Instruct
14
+ library_name: comfyui
15
+ ---
16
+
17
+ # ClipProj v3.1 — measured against the 32B's own variance
18
+
19
+ **Four projection matrices that let a Qwen3-VL-4B or 8B replace the Qwen3-VL-32B text encoder of MiniMax H3.**
20
+
21
+ **15.0 GB → 4.6 GB**, with no change to the diffusion model, the VAEs or the sampler.
22
+
23
+ This release is not about a new architecture. It is about finally knowing **how good these things are**, because the previous numbers could not tell me. This card is mostly the measurement, and the measurement changed three of my own conclusions.
24
+
25
+ Requires the custom node: **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
26
+
27
+ ---
28
+
29
+ ## Read this before the tables: these are metrics, not verdicts
30
+
31
+ **I do not speak these eleven languages.** I cannot tell you whether a render sounds right, and I have not
32
+ asked anyone who can. No native speaker has listened to any of the 297 speech renders on this page.
33
+
34
+ So nothing below is a judgement of quality. Every figure is a **distance between two automatic
35
+ transcriptions** — what one machine wrote down from the reference, against what it wrote down from the
36
+ projection. That is all it is, and it is worth being explicit about what that does and does not capture:
37
+
38
+ **What the numbers do capture.** Whether the same words and the same sounds come out. Two instruments are
39
+ used precisely because each is wrong in a known direction: **Whisper** has a language model inside and
40
+ corrects a slurred word into the most probable real one, so it *under*-reports pronunciation defects — a
41
+ lower bound. **ZIPA** has no lexical decoder at all and counts every shift in realisation as an error, so
42
+ it *over*-reports — an upper bound. What a listener would notice lies between them, and neither number
43
+ alone is the answer.
44
+
45
+ **What they do not capture.** Prosody, rhythm, timbre, naturalness — everything that makes speech sound
46
+ native rather than merely correct. A render scoring 98.8 here could still sound foreign to someone who
47
+ speaks the language. These metrics cannot see that, and neither can I.
48
+
49
+ **So read a score as "close to the 32B, according to this instrument"** — never as "good". If you speak
50
+ one of these languages, your ear outranks every table below, and I would genuinely like to hear what it
51
+ tells you.
52
+
53
+ ---
54
+
55
+ ## What changed in v3.1: giving every script its share
56
+
57
+ v3 was calibrated on a corpus that was overwhelmingly English, with the other languages bolted on
58
+ afterwards as a top-up. v3.1 **adds text and tagged prompts until every writing system carries roughly
59
+ comparable weight** — English excepted, because the prompt format itself is English: the sections, the
60
+ tags and the descriptions are all written in it, so it stays the majority no matter what.
61
+
62
+ Measured share of the v3.1 corpus:
63
+
64
+ | Script | Languages | Tagged prompts | Raw text | Share |
65
+ |---|---|---|---|---|
66
+ | Latin — base | English: the original corpus, image lots and register lots | *base* | — | **68.3 %** |
67
+ | Han | zh | ✓ | ✓ | 6.7 % |
68
+ | Hangul | ko | ✓ | ✓ | 6.6 % |
69
+ | Latin, accented | fr | ✓ | ✓ | 6.3 % |
70
+ | Arabic | ar | ✓ | ✓ **(new)** | 4.1 % |
71
+ | Latin | es, de, it, pt | ✓ | — | 1.3 % each |
72
+ | Cyrillic | ru | ✓ | — | 1.3 % |
73
+
74
+ The rule behind those numbers: **a script that inherits nothing from Latin needs raw text**; a Latin
75
+ script only needs tagged prompts, because the alphabet is already covered and roughly 250 tags are
76
+ enough to attach a language to it. That is why Chinese, Korean and French carry raw lots and Spanish
77
+ does not.
78
+
79
+ **The one genuinely new lot is raw Arabic** — 550 000 characters. Arabic was the last non-Latin script
80
+ still living on tagged prompts alone.
81
+
82
+ The training itself was also restarted from scratch rather than topped up. A network keeps the order it
83
+ learned in: whatever comes last weighs more, and lowering the learning rate on a top-up run does not
84
+ remove that imbalance, it only arbitrates between preserving what was acquired and correcting it. Same
85
+ architecture and same hyper-parameters as v3 — `hidden 32768`, `depth 1`, `tap 24`, `lr 1e-3`, no linear
86
+ path — so what the benchmark below compares is the corpus, not the recipe.
87
+
88
+ ### What it buys
89
+
90
+ Phoneme errors against the 32B, averaged over the four files of each generation, the three seeds and
91
+ compared against the threshold:
92
+
93
+ | | threshold | v3 | **v3.1** | |
94
+ |---|---|---|---|---|
95
+ | es | 0.0 | 1.9 | **0.5** | −74 % |
96
+ | de | 2.7 | 9.8 | **3.6** | −64 % |
97
+ | fr | 4.0 | 8.2 | **3.0** | −64 % |
98
+ | it | 2.3 | 4.2 | **1.6** | −63 % |
99
+ | ru | 10.7 | 21.5 | **13.6** | −37 % |
100
+ | ar | 6.3 | 11.1 | **7.6** | −32 % |
101
+ | zh | 3.3 | 3.0 | **2.2** | −25 % |
102
+ | ja | 6.7 | 7.5 | **6.4** | −14 % |
103
+ | ko | 11.3 | 12.8 | **11.5** | −10 % |
104
+ | pt | 17.0 | 25.1 | **24.1** | −4 % |
105
+ | en | 0.0 | **0.2** | 0.5 | +0.3 |
106
+ | **mean** | 5.8 | **9.6** | **6.8** | **−29 %** |
107
+
108
+ The European languages, which v3 only ever saw as a top-up, gain 60 to 74 %. Arabic gains 32 %, which is
109
+ where the new raw-text lot shows up. Russian gains 37 %.
110
+
111
+ **Raw text does not predict the outcome.** Rapported to each language's own threshold, the two groups
112
+ overlap completely:
113
+
114
+ | | script | raw text | threshold | v3 | v3.1 | × threshold |
115
+ |---|---|---|---|---|---|---|
116
+ | zh | Han | ✓ | 3.3 | 3.0 | 2.2 | **0.67** |
117
+ | it | Latin | — | 2.3 | 4.2 | 1.6 | **0.68** |
118
+ | fr | Latin | ✓ | 4.0 | 8.2 | 3.0 | 0.75 |
119
+ | ja | Kana/Kanji | — | 6.7 | 7.5 | 6.4 | 0.96 |
120
+ | ko | Hangul | ✓ | 11.3 | 12.8 | 11.5 | 1.01 |
121
+ | ar | Arabic | ✓ | 6.3 | 11.1 | 7.6 | 1.20 |
122
+ | ru | Cyrillic | — | 10.7 | 21.5 | 13.6 | 1.27 |
123
+ | de | Latin | — | 2.7 | 9.8 | 3.6 | 1.34 |
124
+ | pt | Latin | — | 17.0 | 25.1 | 24.1 | 1.42 |
125
+
126
+ Languages with a raw lot run 0.67 to 1.20; languages without run 0.68 to 1.42. Italian, with tagged
127
+ prompts only, lands second best overall.
128
+
129
+ **Russian is the clearest case.** It is the only non-Latin script here with no raw text and nothing to
130
+ inherit — Japanese borrows kanji from the Chinese lots, Latin scripts borrow the alphabet from English —
131
+ and it still gains **37 %** between v3 and v3.1 on tagged prompts alone, finishing ahead of German and
132
+ Portuguese, which are Latin. The rule written in the build scripts — *250 tags are enough once the
133
+ alphabet is covered* — evidently extends to Cyrillic, which the Qwen3-VL tokenizer covers natively.
134
+
135
+ So raw text is what an **unseen script** needs, not what a language needs. Where a script is already in
136
+ the tokenizer's reach, tags carry it.
137
+
138
+ **English pays for it, and the bill is half a phoneme out of 77.** That is the whole cost of rebalancing:
139
+ v3 was 0.2 errors, v3.1 is 0.5, both far below anything audible and below what a single seed resolves.
140
+ Portuguese barely moves, but nothing moves in Portuguese — the reference itself scatters by 17 there.
141
+
142
+ The net effect is a change of category rather than a better score. v3 sits at **1.46 to 1.77 times** the
143
+ threshold; v3.1 sits at **1.09 to 1.20**. From measurably worse than a seed change, to indistinguishable
144
+ from one.
145
+
146
+ ---
147
+
148
+ ## The problem with every number I published before
149
+
150
+ A cosine of 0.79, or "23 character errors out of 869" — neither has a scale. Is 23 good? Compared to what? Zero errors is not the right target either, because **the 32B does not reproduce itself**. Change nothing but the seed and it re-pronounces the sentence differently.
151
+
152
+ So the reference is not perfection. It is the 32B compared to itself, same prompt, different seed:
153
+
154
+ | | 32B against itself |
155
+ |---|---|
156
+ | Speech | **5.8 phonemes out of 75** (7.8 %) |
157
+ | Image | **0.9552** SigLIP2 cosine (floor: 0.5313) |
158
+
159
+ That gap is the unit. Everything below is normalised so that **32B = 100**:
160
+
161
+ - **100** — swapping the encoder moves the output as much as changing the seed
162
+ - **above 100** — it moves it less
163
+ - **below 100** — it moves it more
164
+
165
+ Below that threshold you are no longer measuring the projection. You are measuring the generator.
166
+
167
+ ---
168
+
169
+ ## Files
170
+
171
+ Put them in `ComfyUI/models/clip_projections/`.
172
+
173
+ | File | Encoder | Head | Size | Encoder + projection |
174
+ |---|---|---|---|---|
175
+ | `mmh3-4b-ClipProj-v3.1` | any Qwen3-VL-4B | ridge | 26 MB | **4.6 GB** |
176
+ | `mmh3-4b-ClipProj-v3.1-mlp` | any Qwen3-VL-4B | ridge + residual | 481 MB | 5.1 GB |
177
+ | `mmh3-8b-ClipProj-v3.1` | any Qwen3-VL-8B | ridge | 41 MB | 9.6 GB |
178
+ | `mmh3-8b-ClipProj-v3.1-mlp` | any Qwen3-VL-8B | ridge + residual | 577 MB | 10.1 GB |
179
+
180
+ The 8B matrices expect 4096 input dimensions instead of 2560; the node checks the width and refuses a mismatch.
181
+
182
+ ---
183
+
184
+ ## The benchmark
185
+
186
+ Nothing here is a single render. Every figure comes from **three seeds — 42, 100 000 and 100 000 000** — chosen far apart so no one can suspect they are correlated.
187
+
188
+ | | volume |
189
+ |---|---|
190
+ | Speech | 297 renders — 9 conditionings × 11 languages × 3 seeds |
191
+ | Image | 405 renders — 9 conditionings × 15 prompts × 3 seeds |
192
+
193
+ **Speech** is scored in phonemes, by [ZIPA-CR-large](https://huggingface.co/anyspeech/zipa-large-crctc-ns-800k) (88 languages, no lexical decoder — it will not silently repair a botched syllable into a real word), against the 32B **of the same seed**. Distances are Levenshtein throughout.
194
+
195
+ **Image** is scored by SigLIP2 so400m, both against the 32B's render and against the prompt itself — on that second axis the 32B is just one column among nine.
196
+
197
+ Languages: en, fr, es, de, it, pt, ru, ar, zh, ja, ko.
198
+
199
+ ---
200
+
201
+ ## Results
202
+
203
+ | Conditioning | Speech | ± | Image | Prompt |
204
+ |---|---|---|---|---|
205
+ | **32B** *(reference)* | **100.0** | — | **100.0** | **100.0** |
206
+ | `8b-ClipProj-v3.1` | **98.8** | ±1.4 | 100.0 | 100.4 |
207
+ | `8b-ClipProj-v3.1-mlp` | 98.2 | ±1.2 | 101.4 | **102.3** |
208
+ | `4b-ClipProj-v3.1` | 97.9 | ±1.0 | 100.1 | 99.4 |
209
+ | `4b-ClipProj-v3.1-mlp` | 97.8 | ±1.1 | 100.9 | 100.3 |
210
+ | *v3 ridge / mlp, 4B and 8B* | *93.2 – 95.6* | | *99.4 – 102.4* | *99.0 – 100.7* |
211
+
212
+ `±` is the spread of the score across the three seeds. **Two models separated by less than that are not separated at all.**
213
+
214
+ ### The raw counts behind the speech score
215
+
216
+ Three metrics, three units, never added together. **PER** counts phonemes over three seeds on the current
217
+ protocol; **WER** and **CER** count words and characters as Whisper hears them, single-seed on the earlier
218
+ 0.3 MP protocol. They are listed side by side because they disagree in useful ways — see the language
219
+ breakdown below.
220
+
221
+ | | **PER** (3 seeds) | | **WER** (1 seed) | **CER** (1 seed) |
222
+ |---|---|---|---|---|
223
+ | **32B** *(reference)* | **0 / 2469** | — | 6 / 174 | 6 / 869 |
224
+ | *32B against itself* | *~193 / 2469* | *7.8 %* | — | — |
225
+ | `8b-v3.1` | **211 / 2469** | 8.5 % | **10 / 174** | **14 / 869** |
226
+ | `8b-v3.1-mlp` | 222 / 2469 | 9.0 % | 15 / 174 | 28 / 869 |
227
+ | `4b-v3.1-mlp` | 230 / 2469 | 9.3 % | 13 / 174 | 23 / 869 |
228
+ | `4b-v3.1` | 232 / 2469 | 9.4 % | 15 / 174 | 30 / 869 |
229
+ | `4b-v3-mlp` | 281 / 2469 | 11.4 % | 18 / 174 | 32 / 869 |
230
+ | `8b-v3-mlp` | 317 / 2469 | 12.8 % | 17 / 174 | 29 / 869 |
231
+ | `8b-v3` | 326 / 2469 | 13.2 % | 23 / 174 | 46 / 869 |
232
+ | `4b-v3` | 341 / 2469 | 13.8 % | 19 / 174 | 36 / 869 |
233
+
234
+ The 32B scores 0 on PER by construction — it *is* the reference. The row below it is the meaningful one:
235
+ compared to **itself** on another seed it drifts by about 7.8 %, and the four v3.1 files sit at 8.5 to
236
+ 9.4 %. The v3 files sit at 11.4 to 13.8 %, clear of that band.
237
+
238
+ WER and CER rank the files in nearly the same order, which is the point of quoting both: `8b-v3.1` leads
239
+ all three metrics, and no v3 file beats any v3.1 file on any of them.
240
+
241
+ ### Why some scores exceed 100 — and why that is not "better than the 32B"
242
+
243
+ The two image columns do not share a reference, and neither exceedance means what it looks like.
244
+
245
+ **Prompt.** This axis is `cos(image embedding, prompt embedding)`. The 32B is **not** the reference here —
246
+ it is one column among nine, and its value is set to 100 only to give the scale a fixed point. Nothing
247
+ requires it to be the best, and it demonstrably is not: on the prompt asking for a loaf **cut in two**, it
248
+ renders a single piece on two seeds out of three. A projection that follows the description more closely
249
+ earns a higher cosine, legitimately.
250
+
251
+ **Image.** Here 100 *is* the 32B against itself, but the comparison is asymmetric: the threshold pits
252
+ `32B(seed 42)` against `32B(seed 7391)` — two different draws — while a projection is compared to
253
+ `32B(seed 42)`, the **same** draw. It plays with its reference's seed, so the draw noise is removed on its
254
+ side. A score of 101.4 says only *closer to that 32B render than two 32B renders are to each other*.
255
+
256
+ **And none of it is significant.** The nine models span 0.0048 of cosine on the prompt axis, against a
257
+ within-model standard deviation of 0.024 to 0.029 — five times larger. The paired test over 45 cases calls
258
+ all eight projections indistinguishable from the 32B, including the one at 102.3. The +2.3 % is real as a
259
+ measurement and void as a result.
260
+
261
+ ### What actually separates
262
+
263
+ **The corpus, not the size and not the head.** All four v3.1 land within one point of each other — 97.8 to 98.8 — while ranging from 4.6 to 10.1 GB. All four v3 sit a clear notch below, 93.2 to 95.6, at identical sizes. A 4B v3.1 beats an 8B v3 by four points while weighing half as much.
264
+
265
+ **Nothing separates in image.** All nine conditionings, v3 included, are at or above the threshold: 99.4 to 102.4. Swapping the 32B for a 4B changes the picture **less than changing the seed does**. On this axis the 32B is not a ceiling — `8b-v3.1-mlp` scores 102.3 for prompt fidelity, and on one prompt asking for a loaf cut in two, the 32B rendered a single piece on two seeds out of three while the 8B ridge rendered two on all three.
266
+
267
+ **4B against 8B does not separate on general pronunciation.** 97.8 against 98.8, for a seed-to-seed spread of ±1.0 to ±1.4. If you need one number: they are the same.
268
+
269
+ **Ridge against MLP does not separate either.** The residual buys nothing measurable in speech. It shows up in image prompt fidelity — 102.3 against 100.4 on the 8B — but that axis has its own noise and I would not choose a file on it.
270
+
271
+ ### The image measurements in full
272
+
273
+ Two independent SigLIP2 so400m readings over the same 405 renders. First, resemblance to the 32B's own render, same prompt and same seed — 45 cases per model:
274
+
275
+ | | cosine | std. dev. | % of threshold | worst prompt |
276
+ |---|---|---|---|---|
277
+ | **32B against itself** | **0.9552** | — | **100.0** | — |
278
+ | `8b-v3-mlp` | 0.9653 | 0.0328 | 102.4 | 0.8804 |
279
+ | `8b-v3.1-mlp` | 0.9612 | 0.0363 | 101.4 | 0.8808 |
280
+ | `8b-v3` | 0.9594 | 0.0350 | 101.0 | 0.8910 |
281
+ | `4b-v3.1-mlp` | 0.9590 | 0.0321 | 100.9 | 0.9038 |
282
+ | `4b-v3-mlp` | 0.9588 | 0.0408 | 100.8 | 0.8989 |
283
+ | `4b-v3.1` | 0.9557 | 0.0445 | 100.1 | 0.8766 |
284
+ | `8b-v3.1` | 0.9552 | 0.0448 | 100.0 | 0.8893 |
285
+ | `4b-v3` | 0.9528 | 0.0423 | 99.4 | 0.8861 |
286
+
287
+ *Floor: 0.5313 — two 32B renders sharing no content at all still score that, on style and generator artefacts alone.*
288
+
289
+ **The standard deviation settles it.** It runs 0.032 to 0.045, while the entire spread from best to worst model is 0.0125. The scatter within one model is three to four times the gap between models. Nothing here is a ranking.
290
+
291
+ Second, fidelity to the written prompt — an axis where the 32B is one column among nine rather than the reference:
292
+
293
+ | | cosine | std. dev. | base 100 |
294
+ |---|---|---|---|
295
+ | `8b-v3.1-mlp` | 0.1504 | 0.0252 | **102.3** |
296
+ | `4b-v3-mlp` | 0.1481 | 0.0275 | 100.7 |
297
+ | `8b-v3.1` | 0.1477 | 0.0290 | 100.4 |
298
+ | `4b-v3.1-mlp` | 0.1475 | 0.0249 | 100.3 |
299
+ | **32B** | 0.1471 | 0.0245 | **100.0** |
300
+ | `4b-v3.1` | 0.1462 | 0.0272 | 99.4 |
301
+ | `8b-v3` | 0.1460 | 0.0267 | 99.3 |
302
+ | `8b-v3-mlp` | 0.1458 | 0.0236 | 99.2 |
303
+ | `4b-v3` | 0.1456 | 0.0268 | 99.0 |
304
+
305
+ *Floor: −0.0293 — one scene's image against another scene's prompt.*
306
+
307
+ Same verdict, and harder: the spread across all nine models is 0.0048 for a standard deviation of 0.024 to 0.029, **five times larger**. Four models sit above the 32B and four below, in an order that carries no information.
308
+
309
+ The two image axes do not even agree with each other: `8b-v3-mlp` tops the resemblance table and sits second from last on prompt fidelity. Imitating the 32B and following the prompt are not the same objective — the 32B itself misses prompts.
310
+
311
+ ---
312
+
313
+ ## What counts as an error
314
+
315
+ One error is **one phoneme inserted, deleted or substituted** relative to what the 32B pronounced — same prompt, same seed. Levenshtein distance, nothing weighted, nothing forgiven.
316
+
317
+ There is no dictionary in the loop. ZIPA transcribes sound to IPA and has no lexical decoder, so it will not quietly repair a botched syllable into a real word the way a speech-to-text engine would. What it writes down is what came out of the speaker.
318
+
319
+ Concretely, on the French line *"la lumière de Marseille"*, seed 42 — the phonemes following `d ɛ` ("de"):
320
+
321
+ | | | heard as | errors on the line |
322
+ |---|---|---|---|
323
+ | **32B** | `m a ʀ s ɛ j` | *Marseille* | — |
324
+ | `8b-v3.1` | `m a ʀ s ɛ j` | *Marseille* | 4 / 71 |
325
+ | `4b-v3.1` | `m a ʀ s ɛ ʀ ɛ` | *"marcerre"* | 5 / 71 |
326
+ | `4b-v3` | `m a z ɛ ʀ` | *"mazer"* | 8 / 71 |
327
+
328
+ Note how little the toponym costs: `4b-v3.1` botches the name outright and pays **one** phoneme more than `8b-v3.1` over the whole sentence. That is exactly why the aggregate scores cannot settle the proper-noun question, and why it gets its own section below rather than a place in the ranking.
329
+
330
+ ## Per language, because the average hides everything
331
+
332
+ Errors are counted against the 32B **of the same seed**, averaged over the three seeds. The first two columns are the yardstick: how long the reference is, and how much the 32B differs from *itself*.
333
+
334
+ | | length | **threshold** | `8b-v3.1` | `8b-v3.1-mlp` | `4b-v3.1` | `4b-v3.1-mlp` |
335
+ |---|---|---|---|---|---|---|
336
+ | en | 77 | **0.0** | 0.7 | 0.7 | 0.7 | 0.0 |
337
+ | es | 62 | **0.0** | 1.3 | 0.0 | 0.7 | 0.0 |
338
+ | de | 97 | 2.7 | 3.0 | 5.0 | 3.7 | 2.7 |
339
+ | it | 62 | 2.3 | 0.7 | 0.3 | 4.3 | 1.0 |
340
+ | zh | 85 | 3.3 | 3.3 | 1.0 | 2.7 | 2.0 |
341
+ | fr | 70 | 4.0 | 1.7 | 1.3 | 5.0 | 4.0 |
342
+ | ar | 94 | 6.3 | 5.7 | 4.7 | 7.7 | 12.3 |
343
+ | ja | 73 | 6.7 | 6.3 | 6.7 | 5.7 | 7.0 |
344
+ | ru | 77 | 10.7 | 14.0 | 16.3 | 14.0 | 10.0 |
345
+ | ko | 64 | 11.3 | 11.0 | 15.0 | 10.7 | 9.3 |
346
+ | pt | 62 | **17.0** | 22.7 | 23.0 | 22.3 | 28.3 |
347
+
348
+ Read it against the threshold column, never in absolute terms:
349
+
350
+ - **English and Spanish** — the 32B repeats itself phoneme for phoneme. There, a single phoneme of drift is real signal, and all four files stay within one.
351
+ - **French, Italian, Chinese, Arabic, Japanese** — the projections are *at or below* the 32B's own variance. In French both 8B files land at 1.7 and 1.3 against a threshold of 4.0: closer to the 32B than the 32B is to itself.
352
+ - **Portuguese, Russian and Korean** carry thresholds of 17.0, 10.7 and 11.3 — the reference rewrites a large share of its own pronunciation between seeds. Any single-seed comparison there was measuring the dice.
353
+
354
+ ### Where the phoneme metric misleads, and the cross-check that catches it
355
+
356
+ A high threshold does not mean the speech is bad. It means **the phoneme transcriber cannot hold that
357
+ language still.** Cross-checking against Whisper, which reads words rather than sounds, on the same
358
+ renders:
359
+
360
+ | | ZIPA threshold | ZIPA v3.1 | × threshold | **Whisper, 32B** | **Whisper, v3.1** |
361
+ |---|---|---|---|---|---|
362
+ | pt | 17.0 | 24.1 | **1.42** | **0 / 88** | **1.8 / 88** |
363
+ | ru | 10.7 | 13.6 | 1.27 | **0 / 85** | 2.8 / 85 |
364
+ | ko | 11.3 | 11.5 | 1.01 | 2 / 85 | 3.2 / 85 |
365
+ | ja | 6.7 | 6.4 | 0.96 | 2 / 39 | 5.2 / 39 |
366
+ | zh | 3.3 | 2.2 | 0.67 | 2 / 31 | 3.5 / 31 |
367
+
368
+ **Portuguese is the worst language by phoneme and one of the best by word** — zero character errors for
369
+ the 32B, 2 % for the v3.1 files. Russian likewise: Whisper transcribes the 32B and two of the projections
370
+ word for word.
371
+
372
+ The cause is exactly what makes ZIPA useful elsewhere: it has no lexical decoder. European Portuguese
373
+ elides and reduces its vowels, Russian has vowel reduction under stress shift — the phonetic realisation
374
+ moves from one draw to the next while the word does not. ZIPA counts every allophonic variation as an
375
+ error; Whisper, which recognises the word, sees none. In Japanese and Chinese the bias runs the other
376
+ way: Whisper is harsher, because one missed ideogram weighs heavily on 31 characters.
377
+
378
+ **Neither metric is sufficient alone.** Where the ZIPA threshold is high, read the word column.
379
+ (Whisper figures are single-seed, on the earlier 0.3 MP protocol.)
380
+
381
+ ---
382
+
383
+ ## The 32B is one of the least stable models here
384
+
385
+ Distance between two renders of the **same** model, seed changed, nothing else:
386
+
387
+ | | against itself | against the 32B |
388
+ |---|---|---|
389
+ | `4b-v3.1` | **3.6** | 7.0 |
390
+ | `4b-v3.1-mlp` | 4.0 | 7.0 |
391
+ | `8b-v3.1` | 4.1 | 6.4 |
392
+ | `8b-v3.1-mlp` | 4.7 | 6.7 |
393
+ | **32B** | **5.8** | — |
394
+
395
+ The projections repeat themselves *better* than the model they imitate.
396
+
397
+ **And the gap to the 32B is reproducible, not random.** Each projection sits far closer to itself (3.6–4.7) than to the 32B (6.4–7.0). If swapping the encoder merely added randomness, those two columns would match. They do not — each file redoes the same offset on every seed.
398
+
399
+ **It is not an accent either.** An accent would mean one phoneme consistently rendered as another. Counting the actual substitutions says otherwise:
400
+
401
+ | | substitutions | covered by recurring patterns |
402
+ |---|---|---|
403
+ | **32B against itself** | **103** | ɑ→a ×10, ɾ→r ×6, ʒ→ʐ ×5 |
404
+ | `8b-v3.1` | **103** | 4 % — one pattern |
405
+ | `4b-v3.1` | 114 | **0 %** |
406
+ | `8b-v3.1-mlp` | 115 | 8 % |
407
+ | `4b-v3.1-mlp` | 125 | **0 %** |
408
+ | the four v3 files | 153–196 | 2–13 % |
409
+
410
+ The v3.1 files produce **as many substitutions as the 32B inflicts on itself** — 103 to 125 against 103 — and almost none of them form a repeating pattern. The offset is reproducible but scattered across many different sounds rather than concentrated into a signature. Ironically the clearest patterns belong to the 32B itself, between its own seeds, where they are ordinary allophonic variation.
411
+
412
+ The v3 files produce 1.5 to 2 times as many.
413
+
414
+ None of which tells you what it *sounds* like. A native speaker might well hear something none of these counts describe.
415
+
416
+ ---
417
+
418
+ ## An anecdote, and why it is not a result
419
+
420
+ **This measures nothing.** One word, in one language out of eleven, with no denominator — it is recorded
421
+ here because it was noticed, not because it supports a conclusion. It is deliberately absent from every
422
+ table above.
423
+
424
+ The French prompt contains a city name. The four 4B files render it as a non-word on almost every seed,
425
+ the 8B files render it correctly on all three:
426
+
427
+ | | seed 42 | seed 100 k | seed 100 M |
428
+ |---|---|---|---|
429
+ | 32B | ✓ | ✓ | ✓ |
430
+ | `8b-v3.1` | ✓ | ✓ | ✓ |
431
+ | `8b-v3.1-mlp` | ✓ | ✓ | ✓ |
432
+ | `4b-v3.1-mlp` | ✗ | ✗ | ✓ |
433
+ | `4b-v3.1` | ✗ | ✗ | ✗ |
434
+
435
+ What keeps it from being a finding, beyond the sample size: **it costs almost nothing on the sentence.**
436
+ `4b-v3.1` mangles the name outright and ends up **one** phoneme worse than `8b-v3.1` over the whole line.
437
+ Every aggregate metric in this report is blind to it, which cuts both ways — they cannot confirm it either.
438
+
439
+ The aggregate scores say 4B and 8B are equivalent, and that is the conclusion to keep. The only reason to
440
+ mention this at all is that quantisation is independently known to cost factual recall: if your prompts
441
+ lean on names of people or places, test both sizes on **your** prompts rather than trusting anything here.
442
+
443
+ ---
444
+
445
+ ## Which one to take
446
+
447
+ | If you | Take |
448
+ |---|---|
449
+ | generate images or video without speech | **`4b-ClipProj-v3.1`** — 4.6 GB, indistinguishable from the 32B |
450
+ | are tight on VRAM | **`4b-ClipProj-v3.1`** — the ridge is 26 MB and gives up nothing measurable |
451
+ | generate multilingual speech | **`4b-ClipProj-v3.1`** covers nine of the eleven languages tested; `8b-ClipProj-v3.1` has the best overall speech score, by less than the seed-to-seed spread |
452
+ | have the VRAM to spare | `8b-ClipProj-v3.1` — nothing measured says you need it, nothing says it hurts |
453
+
454
+ **Do not take a v3.** That is the only difference this benchmark resolves cleanly: v3 versus v3.1 is real,
455
+ 4B versus 8B is not, ridge versus residual is not.
456
+
457
+ ---
458
+
459
+ ## How the speech benchmark got affordable
460
+
461
+ The old protocol rendered full 0.3 MP video and threw the picture away. Measured, same prompt and seed:
462
+
463
+ | | time |
464
+ |---|---|
465
+ | 0.3 MP video + audio + previews *(old)* | 80.3 s |
466
+ | same, video VAE and previews removed | 48.7 s |
467
+ | **64×64, 8 steps, audio only** | **12.4 s** |
468
+
469
+ Verified lossless before adopting: **+0.2 dB** across every octave band and **1 phoneme out of 70** for dropping the video decode; 64×64 costs 3 phonemes out of 70 against the full-resolution render.
470
+
471
+ **128×128 was rejected** — it truncates the start of the sentence, exactly the same eight phonemes at 6 steps and at 8. 64×64 does not. Counter-intuitive, reproducible, and the reason the whole benchmark runs at the smaller size.
472
+
473
+ That is what made three seeds across 297 renders possible at all: 31 minutes on two cards instead of six and a half hours.
474
+
475
+ ---
476
+
477
+ ## Limitations
478
+
479
+ **Three seeds fix the order of magnitude of the noise, not its tail.** Any gap under one point of score is not a result.
480
+
481
+ **The cosine is blind to countable attributes.** A whole loaf and a halved loaf, same crust, same paper, same light, give the same vector to the fourth decimal. Image equivalence here means *global appearance*, not attribute-by-attribute conformity.
482
+
483
+ **Speech quality is deliberately poor.** Six to eight steps gives a tinny, canned sound — identically for the 32B, with the same 19 dB dip between 1 and 3 kHz. The benchmark measures **correctness of pronunciation, not fidelity of reproduction**.
484
+
485
+ **The phoneme metric is unreliable in Portuguese, Russian and Korean** — not the speech itself. The
486
+ reference drifts by 17.0, 10.7 and 11.3 phonemes there between seeds, while Whisper transcribes the same
487
+ renders with zero to three character errors. Read the word column in those languages.
488
+
489
+ **Quantisation costs facts.** Known before, still true, and the most likely explanation for the proper-noun gap.
490
+
491
+ ---
492
+
493
+ ## Licence and responsibility
494
+
495
+ MIT, like the node. These matrices are derived from the activations of both models and their legal status is unclear; they are provided as-is, for research.
496
+
497
+ - **Qwen3-VL** — Alibaba, Apache 2.0.
498
+ - **MiniMax H3** — custom licence, read it before any commercial use.
499
+
500
+ Not affiliated with, endorsed by, or connected to Alibaba / Qwen, MiniMax, or Comfy Org. You remain responsible for what you generate.
501
+
502
+ ---
503
+
504
+ ## Credits
505
+
506
+ Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number here was measured on this hardware, never estimated. Where a prediction lost to a measurement, the measurement won and the text was rewritten — which happened three times in this release, the largest being a single-seed ranking of the v3.1 files that dissolved entirely once the threshold was known.
bench3.1/audio/ar/32b_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:57b11418c1f77eae19d2babf89fc1fded73b0e3ef59535892d0f5f0514764d8b
3
+ size 485541
bench3.1/audio/ar/32b_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ab9a979b8139deea2222bb98ed3490893e5202a2cc06c7f42a4c8bc02515c80c
3
+ size 452650
bench3.1/audio/ar/32b_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a3f5f0ef60a068367644c5aa411fcc0a695635cfcc37071aa2f0fa6c3234b8e
3
+ size 488148
bench3.1/audio/ar/4b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5a27b47f79fdd7305286a3b8e07e3bb596758a8ab3e7109571af074a2d8c2e8c
3
+ size 467032
bench3.1/audio/ar/4b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ec82f202d8140f32d473d4090ae6810d48b6d4800c100a87673492eb80ee9962
3
+ size 459884
bench3.1/audio/ar/4b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:94e525bca63fee7add1ae785ed858be21199e940c5dcd86cf732e5e23a2b4e20
3
+ size 458763
bench3.1/audio/ar/4b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b13d863e50eac214ac6c1c57fb4a9ebdf4daffd433e7415a78b2070462e839ea
3
+ size 468000
bench3.1/audio/ar/4b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e11ccff921b504b3343fa24664b8006b41c205af367199515887c6c27e9b7958
3
+ size 470773
bench3.1/audio/ar/4b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:770f05d8a34d893f495874d9d8f3fefaa26af704d044edccc801a0c73d119cae
3
+ size 456558
bench3.1/audio/ar/4b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:34d03be6563937655a212e23f9d07258dc8a8a2aa34cf036ba262d3185ba90b3
3
+ size 477859
bench3.1/audio/ar/4b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:825851a2259a4181377aa0648c25d3f253c4135e55f47b9ce2c525adbc9e89f1
3
+ size 471423
bench3.1/audio/ar/4b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cfd3f9a1e8f3b861503a428f8311ddff68fdbeb1785b007a97be9628fa4f421
3
+ size 486703
bench3.1/audio/ar/4b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:532d9d911f3753587c97c69f22127097c855317fbf4cc3c2610724e65c0a7937
3
+ size 484528
bench3.1/audio/ar/4b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e95fef01962c8155082b879d07bcd981800b0af20e720b210bb44cf3007c402e
3
+ size 466523
bench3.1/audio/ar/4b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3e323f0552b09566fdf7fe7dade15753e8afbc7a6a61699cf80a81b8af64323c
3
+ size 476437
bench3.1/audio/ar/8b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:46230a2d96de6707d170fe600178d98e44cd7bd46d5c199c0b56cab15d55f49b
3
+ size 470495
bench3.1/audio/ar/8b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2e416d4942d8d2f83a401d8432867bc243fbff2a3b3683968576670a36c9966f
3
+ size 459569
bench3.1/audio/ar/8b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9cbc8e090a142f33b66de725af3d9f925ef7c62769cd6701897676ea1955ddd2
3
+ size 465130
bench3.1/audio/ar/8b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9da4ba292d06e36a792ec328bbe89f6eb70769b94f4cdea79552d8dad169e3f3
3
+ size 478546
bench3.1/audio/ar/8b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2733da6c8b9603ea0323811ea81076b1f0dbe6e860d7627d4f9b17a4840f6d0c
3
+ size 457617
bench3.1/audio/ar/8b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8d87110720ddc741a444b4a289b1b49206db72df7a4fd3f7c39de2f877414b6
3
+ size 476442
bench3.1/audio/ar/8b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:809f31b63afdaa6c663a2bd1983a7fcb30849842b245028918faeb8552213e1a
3
+ size 481752
bench3.1/audio/ar/8b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:56f60ee0468fa51640af552377d944f892f3b557de1a4ebdc08be218ff2e785b
3
+ size 469274
bench3.1/audio/ar/8b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac31636b767be1cae7ac384f0f703ad7129b0eb3bc7bed1f5cfd078dce26a47e
3
+ size 491750
bench3.1/audio/ar/8b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:abdd4c7625b756495f85957509af9ce7b46236208c463f1822d7a89f9e6d5b67
3
+ size 482115
bench3.1/audio/ar/8b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fa3763570a075e993ac5d3f31c09c91ef883f5d7e8d23e5c4936646944e55742
3
+ size 463509
bench3.1/audio/ar/8b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7013b4ec3ec203164d0cd018c19b1f1fa58904c72bf9c278b0e2901d34b51b3c
3
+ size 490893
bench3.1/audio/de/32b_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:139088593e3c4a2c31477110c1484c28edff2df0da0e9462aad500a1cc250438
3
+ size 447679
bench3.1/audio/de/32b_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a7aa921e29ad449b432ec3970fdc1f66492f86597a641d1b067c1eaa1bbedf01
3
+ size 424947
bench3.1/audio/de/32b_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:afb1026d4b29af96fa59bebe332a4c2e455f16a4ed23495121525776627b8ce1
3
+ size 457866
bench3.1/audio/de/4b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e457d51449c8269939fcc7ad40f9b0e5ee9371523c6323f1cb055dfbae1064df
3
+ size 416507
bench3.1/audio/de/4b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a1b60cd526837d6829c88b524382f86086c78b382b9add11d3bf2d788e9965f
3
+ size 383260
bench3.1/audio/de/4b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40a145b8efd06cb83c34c11de97d32f516a88cf63c8fa6b70204051af7f56ed7
3
+ size 416276
bench3.1/audio/de/4b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9942736d876dc378d27af4fd30201cbf8abc2b62903d6a15e099c540dd4d87cb
3
+ size 424040
bench3.1/audio/de/4b-v3.1-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:59c34aebb926e9b7b85a93a698adbec0a24e7406ed4fceb30a32934e038bd1ae
3
+ size 409861
bench3.1/audio/de/4b-v3.1-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6d66d5a76ed2b232d768d87dfbcd80394f6762abac040e7ea322d47d87b80381
3
+ size 434627
bench3.1/audio/de/4b-v3.1_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:72637556a5a1dc77aa0bca0e5d3c449a6a6f0da784e6bddd1ea865a6a6a104d6
3
+ size 441075
bench3.1/audio/de/4b-v3.1_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:756e835c499eac85b49680623553ea4bd1da20d234c6678ab76e7d50a373df26
3
+ size 438930
bench3.1/audio/de/4b-v3.1_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:986c23484e2bd194cdee2b3a742c617f1c5ac172ec4930bf7491cc83ceb57c2e
3
+ size 457654
bench3.1/audio/de/4b-v3_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:97788cde2fdae7b3ba5044e8c78d46a080f12ceb1852a042110d062de39a17a9
3
+ size 440045
bench3.1/audio/de/4b-v3_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f70eb8f148e6b419959ecba40610fcfcc9c2ee89eb5ff921fc1e2c464715081
3
+ size 427003
bench3.1/audio/de/4b-v3_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25be0a144b40341e7880ca96cc40d5cdd56088cc7a5126cd786c22520ed6838d
3
+ size 465126
bench3.1/audio/de/8b-v3-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a95292be2bff21e091c641965042d106bac04bb933f76a0679bd809863708a85
3
+ size 425098
bench3.1/audio/de/8b-v3-mlp_s100000000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3350935b192160ff5c724a6bded252c0f69a7d0c2d673f41abd1c21997c916b6
3
+ size 412601
bench3.1/audio/de/8b-v3-mlp_s42.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:475ff84db73b26e93e9c07ce01e92098fb2dab6873f80c7aaef7a96bafc69c6e
3
+ size 464571
bench3.1/audio/de/8b-v3.1-mlp_s100000.flac ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:89d21c09bfb7b348ac03ee8ee316e9b75a12867580b7b8e6919cde8c85017623
3
+ size 407809