Duplicate from NicoLab28/ClipProj-MiniMax-H3
Browse filesCo-authored-by: Lab <NicoLab28@users.noreply.huggingface.co>
This view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +527 -0
- README.md +234 -0
- RELEASE_NOTES_v3.md +160 -0
- bench3.1/README.md +506 -0
- bench3.1/audio/ar/32b_s100000.flac +3 -0
- bench3.1/audio/ar/32b_s100000000.flac +3 -0
- bench3.1/audio/ar/32b_s42.flac +3 -0
- bench3.1/audio/ar/4b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/ar/4b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/ar/4b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/ar/4b-v3.1-mlp_s100000.flac +3 -0
- bench3.1/audio/ar/4b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/ar/4b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/ar/4b-v3.1_s100000.flac +3 -0
- bench3.1/audio/ar/4b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/ar/4b-v3.1_s42.flac +3 -0
- bench3.1/audio/ar/4b-v3_s100000.flac +3 -0
- bench3.1/audio/ar/4b-v3_s100000000.flac +3 -0
- bench3.1/audio/ar/4b-v3_s42.flac +3 -0
- bench3.1/audio/ar/8b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/ar/8b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/ar/8b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/ar/8b-v3.1-mlp_s100000.flac +3 -0
- bench3.1/audio/ar/8b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/ar/8b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/ar/8b-v3.1_s100000.flac +3 -0
- bench3.1/audio/ar/8b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/ar/8b-v3.1_s42.flac +3 -0
- bench3.1/audio/ar/8b-v3_s100000.flac +3 -0
- bench3.1/audio/ar/8b-v3_s100000000.flac +3 -0
- bench3.1/audio/ar/8b-v3_s42.flac +3 -0
- bench3.1/audio/de/32b_s100000.flac +3 -0
- bench3.1/audio/de/32b_s100000000.flac +3 -0
- bench3.1/audio/de/32b_s42.flac +3 -0
- bench3.1/audio/de/4b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/de/4b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/de/4b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/de/4b-v3.1-mlp_s100000.flac +3 -0
- bench3.1/audio/de/4b-v3.1-mlp_s100000000.flac +3 -0
- bench3.1/audio/de/4b-v3.1-mlp_s42.flac +3 -0
- bench3.1/audio/de/4b-v3.1_s100000.flac +3 -0
- bench3.1/audio/de/4b-v3.1_s100000000.flac +3 -0
- bench3.1/audio/de/4b-v3.1_s42.flac +3 -0
- bench3.1/audio/de/4b-v3_s100000.flac +3 -0
- bench3.1/audio/de/4b-v3_s100000000.flac +3 -0
- bench3.1/audio/de/4b-v3_s42.flac +3 -0
- bench3.1/audio/de/8b-v3-mlp_s100000.flac +3 -0
- bench3.1/audio/de/8b-v3-mlp_s100000000.flac +3 -0
- bench3.1/audio/de/8b-v3-mlp_s42.flac +3 -0
- bench3.1/audio/de/8b-v3.1-mlp_s100000.flac +3 -0
.gitattributes
ADDED
|
@@ -0,0 +1,527 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
demo/chess-32b-reference.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
demo/chess-4b-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
demo/chess-4b-ridge.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
demo/chess-8b-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
demo/chess-8b-ridge.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
demo/chess-comparison.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
bench3.1/audio/ar/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
bench3.1/audio/ar/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
bench3.1/audio/ar/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
bench3.1/audio/ar/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
bench3.1/audio/ar/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
bench3.1/audio/ar/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
bench3.1/audio/ar/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
bench3.1/audio/ar/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
bench3.1/audio/ar/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
bench3.1/audio/ar/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
bench3.1/audio/ar/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
bench3.1/audio/ar/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
bench3.1/audio/ar/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
bench3.1/audio/ar/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
bench3.1/audio/ar/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
bench3.1/audio/ar/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 58 |
+
bench3.1/audio/ar/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
bench3.1/audio/ar/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 60 |
+
bench3.1/audio/ar/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 61 |
+
bench3.1/audio/ar/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 62 |
+
bench3.1/audio/ar/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 63 |
+
bench3.1/audio/ar/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 64 |
+
bench3.1/audio/ar/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 65 |
+
bench3.1/audio/ar/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 66 |
+
bench3.1/audio/ar/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 67 |
+
bench3.1/audio/ar/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 68 |
+
bench3.1/audio/ar/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 69 |
+
bench3.1/audio/de/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 70 |
+
bench3.1/audio/de/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 71 |
+
bench3.1/audio/de/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 72 |
+
bench3.1/audio/de/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 73 |
+
bench3.1/audio/de/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 74 |
+
bench3.1/audio/de/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 75 |
+
bench3.1/audio/de/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 76 |
+
bench3.1/audio/de/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 77 |
+
bench3.1/audio/de/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 78 |
+
bench3.1/audio/de/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 79 |
+
bench3.1/audio/de/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 80 |
+
bench3.1/audio/de/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 81 |
+
bench3.1/audio/de/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 82 |
+
bench3.1/audio/de/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 83 |
+
bench3.1/audio/de/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 84 |
+
bench3.1/audio/de/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 85 |
+
bench3.1/audio/de/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 86 |
+
bench3.1/audio/de/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 87 |
+
bench3.1/audio/de/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 88 |
+
bench3.1/audio/de/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 89 |
+
bench3.1/audio/de/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 90 |
+
bench3.1/audio/de/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 91 |
+
bench3.1/audio/de/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 92 |
+
bench3.1/audio/de/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 93 |
+
bench3.1/audio/de/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 94 |
+
bench3.1/audio/de/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 95 |
+
bench3.1/audio/de/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 96 |
+
bench3.1/audio/en/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 97 |
+
bench3.1/audio/en/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 98 |
+
bench3.1/audio/en/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 99 |
+
bench3.1/audio/en/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 100 |
+
bench3.1/audio/en/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 101 |
+
bench3.1/audio/en/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 102 |
+
bench3.1/audio/en/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 103 |
+
bench3.1/audio/en/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 104 |
+
bench3.1/audio/en/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 105 |
+
bench3.1/audio/en/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 106 |
+
bench3.1/audio/en/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 107 |
+
bench3.1/audio/en/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 108 |
+
bench3.1/audio/en/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 109 |
+
bench3.1/audio/en/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 110 |
+
bench3.1/audio/en/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 111 |
+
bench3.1/audio/en/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 112 |
+
bench3.1/audio/en/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 113 |
+
bench3.1/audio/en/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 114 |
+
bench3.1/audio/en/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 115 |
+
bench3.1/audio/en/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 116 |
+
bench3.1/audio/en/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 117 |
+
bench3.1/audio/en/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 118 |
+
bench3.1/audio/en/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 119 |
+
bench3.1/audio/en/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 120 |
+
bench3.1/audio/en/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 121 |
+
bench3.1/audio/en/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 122 |
+
bench3.1/audio/en/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 123 |
+
bench3.1/audio/es/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 124 |
+
bench3.1/audio/es/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 125 |
+
bench3.1/audio/es/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 126 |
+
bench3.1/audio/es/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 127 |
+
bench3.1/audio/es/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 128 |
+
bench3.1/audio/es/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 129 |
+
bench3.1/audio/es/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 130 |
+
bench3.1/audio/es/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 131 |
+
bench3.1/audio/es/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 132 |
+
bench3.1/audio/es/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 133 |
+
bench3.1/audio/es/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 134 |
+
bench3.1/audio/es/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 135 |
+
bench3.1/audio/es/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 136 |
+
bench3.1/audio/es/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 137 |
+
bench3.1/audio/es/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 138 |
+
bench3.1/audio/es/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 139 |
+
bench3.1/audio/es/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 140 |
+
bench3.1/audio/es/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 141 |
+
bench3.1/audio/es/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 142 |
+
bench3.1/audio/es/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 143 |
+
bench3.1/audio/es/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 144 |
+
bench3.1/audio/es/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 145 |
+
bench3.1/audio/es/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 146 |
+
bench3.1/audio/es/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 147 |
+
bench3.1/audio/es/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 148 |
+
bench3.1/audio/es/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 149 |
+
bench3.1/audio/es/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 150 |
+
bench3.1/audio/fr/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 151 |
+
bench3.1/audio/fr/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 152 |
+
bench3.1/audio/fr/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 153 |
+
bench3.1/audio/fr/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 154 |
+
bench3.1/audio/fr/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 155 |
+
bench3.1/audio/fr/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 156 |
+
bench3.1/audio/fr/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 157 |
+
bench3.1/audio/fr/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 158 |
+
bench3.1/audio/fr/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 159 |
+
bench3.1/audio/fr/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 160 |
+
bench3.1/audio/fr/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 161 |
+
bench3.1/audio/fr/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 162 |
+
bench3.1/audio/fr/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 163 |
+
bench3.1/audio/fr/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 164 |
+
bench3.1/audio/fr/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 165 |
+
bench3.1/audio/fr/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 166 |
+
bench3.1/audio/fr/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 167 |
+
bench3.1/audio/fr/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 168 |
+
bench3.1/audio/fr/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 169 |
+
bench3.1/audio/fr/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 170 |
+
bench3.1/audio/fr/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 171 |
+
bench3.1/audio/fr/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 172 |
+
bench3.1/audio/fr/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 173 |
+
bench3.1/audio/fr/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 174 |
+
bench3.1/audio/fr/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 175 |
+
bench3.1/audio/fr/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 176 |
+
bench3.1/audio/fr/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 177 |
+
bench3.1/audio/it/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 178 |
+
bench3.1/audio/it/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 179 |
+
bench3.1/audio/it/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 180 |
+
bench3.1/audio/it/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 181 |
+
bench3.1/audio/it/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 182 |
+
bench3.1/audio/it/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 183 |
+
bench3.1/audio/it/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 184 |
+
bench3.1/audio/it/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 185 |
+
bench3.1/audio/it/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 186 |
+
bench3.1/audio/it/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 187 |
+
bench3.1/audio/it/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 188 |
+
bench3.1/audio/it/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 189 |
+
bench3.1/audio/it/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 190 |
+
bench3.1/audio/it/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 191 |
+
bench3.1/audio/it/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 192 |
+
bench3.1/audio/it/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 193 |
+
bench3.1/audio/it/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 194 |
+
bench3.1/audio/it/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 195 |
+
bench3.1/audio/it/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 196 |
+
bench3.1/audio/it/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 197 |
+
bench3.1/audio/it/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 198 |
+
bench3.1/audio/it/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 199 |
+
bench3.1/audio/it/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 200 |
+
bench3.1/audio/it/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 201 |
+
bench3.1/audio/it/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 202 |
+
bench3.1/audio/it/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 203 |
+
bench3.1/audio/it/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 204 |
+
bench3.1/audio/ja/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 205 |
+
bench3.1/audio/ja/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 206 |
+
bench3.1/audio/ja/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 207 |
+
bench3.1/audio/ja/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 208 |
+
bench3.1/audio/ja/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 209 |
+
bench3.1/audio/ja/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 210 |
+
bench3.1/audio/ja/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 211 |
+
bench3.1/audio/ja/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 212 |
+
bench3.1/audio/ja/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 213 |
+
bench3.1/audio/ja/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 214 |
+
bench3.1/audio/ja/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 215 |
+
bench3.1/audio/ja/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 216 |
+
bench3.1/audio/ja/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 217 |
+
bench3.1/audio/ja/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 218 |
+
bench3.1/audio/ja/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 219 |
+
bench3.1/audio/ja/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 220 |
+
bench3.1/audio/ja/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 221 |
+
bench3.1/audio/ja/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 222 |
+
bench3.1/audio/ja/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 223 |
+
bench3.1/audio/ja/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 224 |
+
bench3.1/audio/ja/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 225 |
+
bench3.1/audio/ja/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 226 |
+
bench3.1/audio/ja/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 227 |
+
bench3.1/audio/ja/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 228 |
+
bench3.1/audio/ja/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 229 |
+
bench3.1/audio/ja/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 230 |
+
bench3.1/audio/ja/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 231 |
+
bench3.1/audio/ko/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 232 |
+
bench3.1/audio/ko/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 233 |
+
bench3.1/audio/ko/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 234 |
+
bench3.1/audio/ko/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 235 |
+
bench3.1/audio/ko/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 236 |
+
bench3.1/audio/ko/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 237 |
+
bench3.1/audio/ko/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 238 |
+
bench3.1/audio/ko/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 239 |
+
bench3.1/audio/ko/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 240 |
+
bench3.1/audio/ko/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 241 |
+
bench3.1/audio/ko/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 242 |
+
bench3.1/audio/ko/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 243 |
+
bench3.1/audio/ko/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 244 |
+
bench3.1/audio/ko/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 245 |
+
bench3.1/audio/ko/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 246 |
+
bench3.1/audio/ko/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 247 |
+
bench3.1/audio/ko/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 248 |
+
bench3.1/audio/ko/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 249 |
+
bench3.1/audio/ko/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 250 |
+
bench3.1/audio/ko/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 251 |
+
bench3.1/audio/ko/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 252 |
+
bench3.1/audio/ko/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 253 |
+
bench3.1/audio/ko/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 254 |
+
bench3.1/audio/ko/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 255 |
+
bench3.1/audio/ko/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 256 |
+
bench3.1/audio/ko/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 257 |
+
bench3.1/audio/ko/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 258 |
+
bench3.1/audio/pt/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 259 |
+
bench3.1/audio/pt/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 260 |
+
bench3.1/audio/pt/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 261 |
+
bench3.1/audio/pt/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 262 |
+
bench3.1/audio/pt/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 263 |
+
bench3.1/audio/pt/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 264 |
+
bench3.1/audio/pt/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 265 |
+
bench3.1/audio/pt/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 266 |
+
bench3.1/audio/pt/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 267 |
+
bench3.1/audio/pt/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 268 |
+
bench3.1/audio/pt/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 269 |
+
bench3.1/audio/pt/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 270 |
+
bench3.1/audio/pt/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 271 |
+
bench3.1/audio/pt/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 272 |
+
bench3.1/audio/pt/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 273 |
+
bench3.1/audio/pt/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 274 |
+
bench3.1/audio/pt/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 275 |
+
bench3.1/audio/pt/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 276 |
+
bench3.1/audio/pt/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 277 |
+
bench3.1/audio/pt/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 278 |
+
bench3.1/audio/pt/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 279 |
+
bench3.1/audio/pt/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 280 |
+
bench3.1/audio/pt/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 281 |
+
bench3.1/audio/pt/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 282 |
+
bench3.1/audio/pt/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 283 |
+
bench3.1/audio/pt/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 284 |
+
bench3.1/audio/pt/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 285 |
+
bench3.1/audio/ru/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 286 |
+
bench3.1/audio/ru/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 287 |
+
bench3.1/audio/ru/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 288 |
+
bench3.1/audio/ru/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 289 |
+
bench3.1/audio/ru/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 290 |
+
bench3.1/audio/ru/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 291 |
+
bench3.1/audio/ru/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 292 |
+
bench3.1/audio/ru/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 293 |
+
bench3.1/audio/ru/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 294 |
+
bench3.1/audio/ru/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 295 |
+
bench3.1/audio/ru/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 296 |
+
bench3.1/audio/ru/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 297 |
+
bench3.1/audio/ru/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 298 |
+
bench3.1/audio/ru/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 299 |
+
bench3.1/audio/ru/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 300 |
+
bench3.1/audio/ru/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 301 |
+
bench3.1/audio/ru/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 302 |
+
bench3.1/audio/ru/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 303 |
+
bench3.1/audio/ru/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 304 |
+
bench3.1/audio/ru/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 305 |
+
bench3.1/audio/ru/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 306 |
+
bench3.1/audio/ru/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 307 |
+
bench3.1/audio/ru/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 308 |
+
bench3.1/audio/ru/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 309 |
+
bench3.1/audio/ru/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 310 |
+
bench3.1/audio/ru/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 311 |
+
bench3.1/audio/ru/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 312 |
+
bench3.1/audio/zh/32b_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 313 |
+
bench3.1/audio/zh/32b_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 314 |
+
bench3.1/audio/zh/32b_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 315 |
+
bench3.1/audio/zh/4b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 316 |
+
bench3.1/audio/zh/4b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 317 |
+
bench3.1/audio/zh/4b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 318 |
+
bench3.1/audio/zh/4b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 319 |
+
bench3.1/audio/zh/4b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 320 |
+
bench3.1/audio/zh/4b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 321 |
+
bench3.1/audio/zh/4b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 322 |
+
bench3.1/audio/zh/4b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 323 |
+
bench3.1/audio/zh/4b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 324 |
+
bench3.1/audio/zh/4b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 325 |
+
bench3.1/audio/zh/4b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 326 |
+
bench3.1/audio/zh/4b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 327 |
+
bench3.1/audio/zh/8b-v3-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 328 |
+
bench3.1/audio/zh/8b-v3-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 329 |
+
bench3.1/audio/zh/8b-v3-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 330 |
+
bench3.1/audio/zh/8b-v3.1-mlp_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 331 |
+
bench3.1/audio/zh/8b-v3.1-mlp_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 332 |
+
bench3.1/audio/zh/8b-v3.1-mlp_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 333 |
+
bench3.1/audio/zh/8b-v3.1_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 334 |
+
bench3.1/audio/zh/8b-v3.1_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 335 |
+
bench3.1/audio/zh/8b-v3.1_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 336 |
+
bench3.1/audio/zh/8b-v3_s100000.flac filter=lfs diff=lfs merge=lfs -text
|
| 337 |
+
bench3.1/audio/zh/8b-v3_s100000000.flac filter=lfs diff=lfs merge=lfs -text
|
| 338 |
+
bench3.1/audio/zh/8b-v3_s42.flac filter=lfs diff=lfs merge=lfs -text
|
| 339 |
+
bench3.1/images/brutes/p01_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 340 |
+
bench3.1/images/brutes/p01_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 341 |
+
bench3.1/images/brutes/p01_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 342 |
+
bench3.1/images/brutes/p01_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 343 |
+
bench3.1/images/brutes/p01_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 344 |
+
bench3.1/images/brutes/p02_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 345 |
+
bench3.1/images/brutes/p02_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 346 |
+
bench3.1/images/brutes/p02_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 347 |
+
bench3.1/images/brutes/p02_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 348 |
+
bench3.1/images/brutes/p02_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 349 |
+
bench3.1/images/brutes/p03_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 350 |
+
bench3.1/images/brutes/p03_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 351 |
+
bench3.1/images/brutes/p03_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 352 |
+
bench3.1/images/brutes/p03_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 353 |
+
bench3.1/images/brutes/p03_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 354 |
+
bench3.1/images/brutes/p04_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 355 |
+
bench3.1/images/brutes/p04_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 356 |
+
bench3.1/images/brutes/p04_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 357 |
+
bench3.1/images/brutes/p04_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 358 |
+
bench3.1/images/brutes/p04_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 359 |
+
bench3.1/images/brutes/p05_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 360 |
+
bench3.1/images/brutes/p05_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 361 |
+
bench3.1/images/brutes/p05_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 362 |
+
bench3.1/images/brutes/p05_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 363 |
+
bench3.1/images/brutes/p05_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 364 |
+
bench3.1/images/brutes/p06_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 365 |
+
bench3.1/images/brutes/p06_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 366 |
+
bench3.1/images/brutes/p06_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 367 |
+
bench3.1/images/brutes/p06_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 368 |
+
bench3.1/images/brutes/p06_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 369 |
+
bench3.1/images/brutes/p07_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 370 |
+
bench3.1/images/brutes/p07_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 371 |
+
bench3.1/images/brutes/p07_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 372 |
+
bench3.1/images/brutes/p07_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 373 |
+
bench3.1/images/brutes/p07_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 374 |
+
bench3.1/images/brutes/p08_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 375 |
+
bench3.1/images/brutes/p08_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 376 |
+
bench3.1/images/brutes/p08_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 377 |
+
bench3.1/images/brutes/p08_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 378 |
+
bench3.1/images/brutes/p08_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 379 |
+
bench3.1/images/brutes/p09_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 380 |
+
bench3.1/images/brutes/p09_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 381 |
+
bench3.1/images/brutes/p09_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 382 |
+
bench3.1/images/brutes/p09_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 383 |
+
bench3.1/images/brutes/p09_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 384 |
+
bench3.1/images/brutes/p10_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 385 |
+
bench3.1/images/brutes/p10_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 386 |
+
bench3.1/images/brutes/p10_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 387 |
+
bench3.1/images/brutes/p10_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 388 |
+
bench3.1/images/brutes/p10_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 389 |
+
bench3.1/images/brutes/p11_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 390 |
+
bench3.1/images/brutes/p11_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 391 |
+
bench3.1/images/brutes/p11_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 392 |
+
bench3.1/images/brutes/p11_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 393 |
+
bench3.1/images/brutes/p11_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 394 |
+
bench3.1/images/brutes/p12_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 395 |
+
bench3.1/images/brutes/p12_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 396 |
+
bench3.1/images/brutes/p12_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 397 |
+
bench3.1/images/brutes/p12_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 398 |
+
bench3.1/images/brutes/p12_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 399 |
+
bench3.1/images/brutes/p13_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 400 |
+
bench3.1/images/brutes/p13_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 401 |
+
bench3.1/images/brutes/p13_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 402 |
+
bench3.1/images/brutes/p13_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 403 |
+
bench3.1/images/brutes/p13_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 404 |
+
bench3.1/images/brutes/p14_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 405 |
+
bench3.1/images/brutes/p14_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 406 |
+
bench3.1/images/brutes/p14_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 407 |
+
bench3.1/images/brutes/p14_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 408 |
+
bench3.1/images/brutes/p14_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 409 |
+
bench3.1/images/brutes/p15_32b.png filter=lfs diff=lfs merge=lfs -text
|
| 410 |
+
bench3.1/images/brutes/p15_4b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 411 |
+
bench3.1/images/brutes/p15_4b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 412 |
+
bench3.1/images/brutes/p15_8b-v3.1-mlp.png filter=lfs diff=lfs merge=lfs -text
|
| 413 |
+
bench3.1/images/brutes/p15_8b-v3.1.png filter=lfs diff=lfs merge=lfs -text
|
| 414 |
+
bench3.1/images/planches/p01.jpg filter=lfs diff=lfs merge=lfs -text
|
| 415 |
+
bench3.1/images/planches/p02.jpg filter=lfs diff=lfs merge=lfs -text
|
| 416 |
+
bench3.1/images/planches/p03.jpg filter=lfs diff=lfs merge=lfs -text
|
| 417 |
+
bench3.1/images/planches/p04.jpg filter=lfs diff=lfs merge=lfs -text
|
| 418 |
+
bench3.1/images/planches/p05.jpg filter=lfs diff=lfs merge=lfs -text
|
| 419 |
+
bench3.1/images/planches/p06.jpg filter=lfs diff=lfs merge=lfs -text
|
| 420 |
+
bench3.1/images/planches/p07.jpg filter=lfs diff=lfs merge=lfs -text
|
| 421 |
+
bench3.1/images/planches/p08.jpg filter=lfs diff=lfs merge=lfs -text
|
| 422 |
+
bench3.1/images/planches/p09.jpg filter=lfs diff=lfs merge=lfs -text
|
| 423 |
+
bench3.1/images/planches/p10.jpg filter=lfs diff=lfs merge=lfs -text
|
| 424 |
+
bench3.1/images/planches/p12.jpg filter=lfs diff=lfs merge=lfs -text
|
| 425 |
+
bench3.1/images/planches/p13.jpg filter=lfs diff=lfs merge=lfs -text
|
| 426 |
+
bench3.1/images/planches/p14.jpg filter=lfs diff=lfs merge=lfs -text
|
| 427 |
+
bench3.1/images/planches/p15.jpg filter=lfs diff=lfs merge=lfs -text
|
| 428 |
+
bench3.1/video/clipproj-v3.1-eleven-languages.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 429 |
+
bench3.1/video/langues/ar/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 430 |
+
bench3.1/video/langues/ar/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 431 |
+
bench3.1/video/langues/ar/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 432 |
+
bench3.1/video/langues/ar/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 433 |
+
bench3.1/video/langues/ar/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 434 |
+
bench3.1/video/langues/ar/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 435 |
+
bench3.1/video/langues/ar/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 436 |
+
bench3.1/video/langues/ar/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 437 |
+
bench3.1/video/langues/ar/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 438 |
+
bench3.1/video/langues/de/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 439 |
+
bench3.1/video/langues/de/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 440 |
+
bench3.1/video/langues/de/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 441 |
+
bench3.1/video/langues/de/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 442 |
+
bench3.1/video/langues/de/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 443 |
+
bench3.1/video/langues/de/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 444 |
+
bench3.1/video/langues/de/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 445 |
+
bench3.1/video/langues/de/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 446 |
+
bench3.1/video/langues/de/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 447 |
+
bench3.1/video/langues/en/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 448 |
+
bench3.1/video/langues/en/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 449 |
+
bench3.1/video/langues/en/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 450 |
+
bench3.1/video/langues/en/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 451 |
+
bench3.1/video/langues/en/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 452 |
+
bench3.1/video/langues/en/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 453 |
+
bench3.1/video/langues/en/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 454 |
+
bench3.1/video/langues/en/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 455 |
+
bench3.1/video/langues/en/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 456 |
+
bench3.1/video/langues/es/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 457 |
+
bench3.1/video/langues/es/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 458 |
+
bench3.1/video/langues/es/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 459 |
+
bench3.1/video/langues/es/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 460 |
+
bench3.1/video/langues/es/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 461 |
+
bench3.1/video/langues/es/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 462 |
+
bench3.1/video/langues/es/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 463 |
+
bench3.1/video/langues/es/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 464 |
+
bench3.1/video/langues/es/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 465 |
+
bench3.1/video/langues/fr/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 466 |
+
bench3.1/video/langues/fr/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 467 |
+
bench3.1/video/langues/fr/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 468 |
+
bench3.1/video/langues/fr/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 469 |
+
bench3.1/video/langues/fr/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 470 |
+
bench3.1/video/langues/fr/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 471 |
+
bench3.1/video/langues/fr/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 472 |
+
bench3.1/video/langues/fr/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 473 |
+
bench3.1/video/langues/fr/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 474 |
+
bench3.1/video/langues/it/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 475 |
+
bench3.1/video/langues/it/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 476 |
+
bench3.1/video/langues/it/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 477 |
+
bench3.1/video/langues/it/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 478 |
+
bench3.1/video/langues/it/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 479 |
+
bench3.1/video/langues/it/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 480 |
+
bench3.1/video/langues/it/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 481 |
+
bench3.1/video/langues/it/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 482 |
+
bench3.1/video/langues/it/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 483 |
+
bench3.1/video/langues/ja/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 484 |
+
bench3.1/video/langues/ja/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 485 |
+
bench3.1/video/langues/ja/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 486 |
+
bench3.1/video/langues/ja/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 487 |
+
bench3.1/video/langues/ja/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 488 |
+
bench3.1/video/langues/ja/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 489 |
+
bench3.1/video/langues/ja/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 490 |
+
bench3.1/video/langues/ja/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 491 |
+
bench3.1/video/langues/ja/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 492 |
+
bench3.1/video/langues/ko/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 493 |
+
bench3.1/video/langues/ko/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 494 |
+
bench3.1/video/langues/ko/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 495 |
+
bench3.1/video/langues/ko/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 496 |
+
bench3.1/video/langues/ko/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 497 |
+
bench3.1/video/langues/ko/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 498 |
+
bench3.1/video/langues/ko/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 499 |
+
bench3.1/video/langues/ko/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 500 |
+
bench3.1/video/langues/ko/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 501 |
+
bench3.1/video/langues/pt/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 502 |
+
bench3.1/video/langues/pt/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 503 |
+
bench3.1/video/langues/pt/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 504 |
+
bench3.1/video/langues/pt/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 505 |
+
bench3.1/video/langues/pt/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 506 |
+
bench3.1/video/langues/pt/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 507 |
+
bench3.1/video/langues/pt/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 508 |
+
bench3.1/video/langues/pt/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 509 |
+
bench3.1/video/langues/pt/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 510 |
+
bench3.1/video/langues/ru/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 511 |
+
bench3.1/video/langues/ru/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 512 |
+
bench3.1/video/langues/ru/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 513 |
+
bench3.1/video/langues/ru/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 514 |
+
bench3.1/video/langues/ru/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 515 |
+
bench3.1/video/langues/ru/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 516 |
+
bench3.1/video/langues/ru/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 517 |
+
bench3.1/video/langues/ru/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 518 |
+
bench3.1/video/langues/ru/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 519 |
+
bench3.1/video/langues/zh/32b.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 520 |
+
bench3.1/video/langues/zh/4b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 521 |
+
bench3.1/video/langues/zh/4b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 522 |
+
bench3.1/video/langues/zh/4b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 523 |
+
bench3.1/video/langues/zh/4b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 524 |
+
bench3.1/video/langues/zh/8b-v3-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 525 |
+
bench3.1/video/langues/zh/8b-v3.1-mlp.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 526 |
+
bench3.1/video/langues/zh/8b-v3.1.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 527 |
+
bench3.1/video/langues/zh/8b-v3.mp4 filter=lfs diff=lfs merge=lfs -text
|
README.md
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: mit
|
| 3 |
+
tags:
|
| 4 |
+
- comfyui
|
| 5 |
+
- minimax-h3
|
| 6 |
+
- text-to-video
|
| 7 |
+
- qwen3-vl
|
| 8 |
+
- text-encoder
|
| 9 |
+
base_model:
|
| 10 |
+
- Comfy-Org/MiniMax-H3
|
| 11 |
+
- Qwen/Qwen3-VL-4B-Instruct
|
| 12 |
+
library_name: comfyui
|
| 13 |
+
---
|
| 14 |
+
|
| 15 |
+
# ClipProj — MiniMax H3 conditioning from a Qwen3-VL-4B or 8B
|
| 16 |
+
|
| 17 |
+
**Projection matrices that let a small Qwen3-VL replace the Qwen3-VL-32B text encoder of MiniMax H3.**
|
| 18 |
+
|
| 19 |
+
**15.7 GB → 4.5 GB of VRAM**, with no change to the diffusion model, the VAEs or the sampler.
|
| 20 |
+
|
| 21 |
+
---
|
| 22 |
+
|
| 23 |
+
## v3.1 — better multilingual speech
|
| 24 |
+
|
| 25 |
+
**No measurable gain in image over v3 — the whole gain is in speech.** The image scores of the two generations overlap entirely, and nothing measured here separates either of them from the 32B, which is not the same as saying they are identical to it: the cosine is blind to countable attributes. What v3.1 improves is pronunciation in the ten languages that are not English — **29 %% fewer phoneme errors overall, 60 to 74 %% fewer in Spanish, French, German and Italian.**
|
| 26 |
+
|
| 27 |
+
<video controls width="360" src="https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4"></video>
|
| 28 |
+
|
| 29 |
+
*Eleven languages, 88 seconds. For each one the **smallest** file that matches the 32B — not the best one. Nine of the eleven run on a 4B. [Direct link](https://huggingface.co/NicoLab28/ClipProj-MiniMax-H3/resolve/main/bench3.1/video/clipproj-v3.1-eleven-languages.mp4)*
|
| 30 |
+
|
| 31 |
+
> ⚠️ **The video looks and sounds rough, on purpose.** 0.3 MP, 6 sampling steps — the settings that made 297 renders affordable — then upscaled. The audio is tinny for the same reason, **identically so on the 32B**, with the same 19 dB dip between 1 and 3 kHz. This is a pronunciation test, not a showcase: it exists to let you hear *which words come out*.
|
| 32 |
+
|
| 33 |
+
**Four new files:** `mmh3-4b-ClipProj-v3.1`, `mmh3-4b-ClipProj-v3.1-mlp`, `mmh3-8b-ClipProj-v3.1`, `mmh3-8b-ClipProj-v3.1-mlp`. They load on node **0.1.13** with no code change — same base as v3. Start with **`mmh3-4b-ClipProj-v3.1`**: 26 MB of projection, 4.6 GB with its encoder.
|
| 34 |
+
|
| 35 |
+
**What changed:** the calibration corpus now gives every writing system a comparable share instead of being overwhelmingly English, with raw Arabic text added. Phoneme errors drop **29 % overall**, 60 to 74 % on the European languages.
|
| 36 |
+
|
| 37 |
+
**What it is measured against:** the 32B does not reproduce itself. Change nothing but the seed and it re-pronounces **5.8 phonemes out of 75** differently. The four v3.1 files sit at 6.4 to 7.0 — so swapping the encoder costs about what re-rolling the seed costs. Every figure is normalised against that variance rather than against zero, and on that scale nothing separates 4B from 8B, or ridge from residual.
|
| 38 |
+
|
| 39 |
+
**Full report** — three seeds, 297 speech renders, 405 image renders, method, limitations and all the raw data: **[bench3.1/README.md](./bench3.1/README.md)**
|
| 40 |
+
|
| 41 |
+
---
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
> ⚠️ **Proof of concept — working, but a proof of concept.** It runs and produces good video, and every number below was measured on real hardware. Built and tested on a single setup (Windows 11, NVIDIA, ComfyUI 0.31.0) with deliberately limited exploration.
|
| 45 |
+
|
| 46 |
+
> **Arriving from a tutorial or an article?** Anything published before 11 August names files that have moved. `h3_qwen3vl_4b_tap24`, `h3_control_zero` and `h3_control_identity` are still here, one folder down in `obsolete/`, so nothing is lost. But take the current set instead: **`mmh3-4b-ClipProj-celeb-mlp`** for a Qwen3-VL-4B, **`mmh3-8b-ClipProj-celeb-mlp`** for an 8B. They are better on every measurement below, and they need node **0.1.4 or later**.
|
| 47 |
+
|
| 48 |
+
These files are useless on their own. They require the custom node:
|
| 49 |
+
**[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
|
| 50 |
+
|
| 51 |
+
## Where this came from
|
| 52 |
+
|
| 53 |
+
I am not an ML researcher. I work in imaging, and programming is a tool and a hobby rather than my trade. This started as something to tinker with: I wanted to understand how a diffusion model actually uses its text encoder, and the only way I know how to understand something is to take it apart and see whether it still runs afterwards.
|
| 54 |
+
|
| 55 |
+
So the question was never "how do I save VRAM". It was "is this even possible at all". I expected it to fail. A linear map between two models that were never trained together, fitted in a single pass with no gradients and no learning rate, has no business producing usable video.
|
| 56 |
+
|
| 57 |
+
It did, and the first results were good enough that keeping them on my own disk seemed silly. That is the whole story, and it is why this is labelled a proof of concept rather than a tool: it was never designed as one.
|
| 58 |
+
|
| 59 |
+
It is also why there are so many measurements on the model card. Before showing this to anyone I had to convince myself I was not fooling myself, and most of what I tried along the way turned out to be wrong. Those attempts are written down as well, in [MEASUREMENTS.md](https://github.com/nicolab28/ComfyUI-ClipProj/blob/main/MEASUREMENTS.md) and [CALIBRATION.md](https://github.com/nicolab28/ComfyUI-ClipProj/blob/main/CALIBRATION.md).
|
| 60 |
+
|
| 61 |
+
## Update to node 0.1.4, and re-download the `-mlp` matrices
|
| 62 |
+
|
| 63 |
+
**The `-mlp` matrices are now fp16 and half the size.** The residual network was published in fp32 and the node forced fp32 on load regardless of the file, so storing it in half precision would have halved the download and saved nothing at all in VRAM. Node 0.1.4 keeps a residual in whatever precision it was saved in, converting its inputs and outputs around it instead. Measured: 240 MB on the card instead of 480 for the 4B, 288 instead of 576 for the 8B. The files here have been replaced under the same names — re-download them, and take 0.1.4 with them, because an older node will load them and cast them straight back up to fp32.
|
| 64 |
+
|
| 65 |
+
Node 0.1.4 also frees the card **before** loading a replacement encoder rather than after, which matters if yours is tight enough that two encoders will not sit on it at once.
|
| 66 |
+
|
| 67 |
+
## Also in 0.1.3
|
| 68 |
+
|
| 69 |
+
Two reasons, one of them silent.
|
| 70 |
+
|
| 71 |
+
**The `-mlp` matrices carry a residual network, and an older node ignores it without saying so.** It reads the matrix, finds keys it does not know, drops them, and applies the linear part alone. Nothing fails, nothing warns, and you end up judging the plain matrix while believing you tested the residual. Node 0.1.3 reads them.
|
| 72 |
+
|
| 73 |
+
**Everything is renamed.** The old `h3_qwen3vl_*` files have moved to `obsolete/` and the `.pt` copies are gone: opening a pickle executes code, which makes no sense for a file holding six tensors. If a workflow of yours names an old file, either point it at `obsolete/` or, better, switch to the new set.
|
| 74 |
+
|
| 75 |
+
## What this is
|
| 76 |
+
|
| 77 |
+
MiniMax H3 conditions on a Qwen3-VL-32B truncated to 50 layers — 15.7 GB in NVFP4 — solely to turn a prompt into a `[seq, 5120]` tensor. This repository provides a learned map that lets a much smaller Qwen3-VL produce the same conditioning:
|
| 78 |
+
|
| 79 |
+
```
|
| 80 |
+
cond = ((h - mean_in) / std_in) @ W * std_out + mean_out
|
| 81 |
+
```
|
| 82 |
+
|
| 83 |
+
and, in the `-mlp` files, plus the output of a small residual network fed the same standardised input.
|
| 84 |
+
|
| 85 |
+
It works because every Qwen3-VL shares the **same tokenizer** (151936 tokens): a prompt yields the same tokens at the same positions in both models, so a position-by-position mapping between their hidden states can be learned. The matrix is fitted by plain **ridge regression** — no gradients, no epochs, no learning rate. The residual network is the only part that is trained.
|
| 86 |
+
|
| 87 |
+
## Files
|
| 88 |
+
|
| 89 |
+
Put them in `ComfyUI/models/clip_projections/`.
|
| 90 |
+
|
| 91 |
+
**Start with `mmh3-8b-ClipProj-celeb-mlp` if you have the VRAM, `mmh3-4b-ClipProj-celeb-mlp` otherwise.**
|
| 92 |
+
|
| 93 |
+
| File | Encoder | Names covered | Residual | Test cosine |
|
| 94 |
+
|---|---|---|---|---|
|
| 95 |
+
| `mmh3-4b-ClipProj` | any Qwen3-VL-4B | no | no | 0.7169 |
|
| 96 |
+
| `mmh3-4b-ClipProj-mlp` | any Qwen3-VL-4B | no | yes | 0.7944 |
|
| 97 |
+
| `mmh3-4b-ClipProj-celeb` | any Qwen3-VL-4B | **yes** | no | 0.7095 |
|
| 98 |
+
| `mmh3-4b-ClipProj-celeb-mlp` | any Qwen3-VL-4B | **yes** | yes | 0.7930 |
|
| 99 |
+
| `mmh3-8b-ClipProj` | any Qwen3-VL-8B | no | no | 0.7528 |
|
| 100 |
+
| `mmh3-8b-ClipProj-mlp` | any Qwen3-VL-8B | no | yes | 0.7970 |
|
| 101 |
+
| `mmh3-8b-ClipProj-celeb` | any Qwen3-VL-8B | **yes** | no | 0.7466 |
|
| 102 |
+
| `mmh3-8b-ClipProj-celeb-mlp` | any Qwen3-VL-8B | **yes** | yes | **0.8037** |
|
| 103 |
+
| `mmh3-ClipProj-control-zero` | — | control, run it once | — | — |
|
| 104 |
+
| `mmh3-ClipProj-control-identity` | — | control, run it once | — | — |
|
| 105 |
+
|
| 106 |
+
All eight are calibrated on the same general corpus and measured on the same held-out prompts, so the column is comparable across every row.
|
| 107 |
+
|
| 108 |
+
Every matrix works on **any variant of its own size**: the measured cosine gap between a bf16-calibrated matrix applied to an abliterated fp8 encoder is 0.0023. You do not need the exact checkpoint a matrix was calibrated on. The 8B matrices need an 8B encoder though — 4096 input dimensions instead of 2560 — and the node checks the width and refuses a mismatch.
|
| 109 |
+
|
| 110 |
+
## Named people
|
| 111 |
+
|
| 112 |
+
**This is what changed in 0.1.3, and it was a corpus problem.**
|
| 113 |
+
|
| 114 |
+
The calibration corpus named a person on about 70 lines out of 8632, roughly 0.02 % of the training tokens. The directions of the hidden space that carry an identity were therefore constrained by nothing at all, and the fit put whatever minimised the error on landscape descriptions there. Named people came out as somebody else.
|
| 115 |
+
|
| 116 |
+
The `-celeb` matrices add 500 people, ranked by popularity, with five short prompts and two long ones each. What it buys and what it costs:
|
| 117 |
+
|
| 118 |
+
| | name tokens | rest of the sentence | general test set |
|
| 119 |
+
|---|---|---|---|
|
| 120 |
+
| without | 0.8265 | 0.9358 | 0.7944 |
|
| 121 |
+
| with | **0.8844** | **0.9516** | 0.7930 |
|
| 122 |
+
|
| 123 |
+
Seven thousandths of cosine on the general corpus, for six points on the tokens that carry an identity. The rest of the sentence improves too, because the celebrity prompts are short and the general corpus had nothing under fifteen words.
|
| 124 |
+
|
| 125 |
+
Two findings that decide how far this is worth pushing.
|
| 126 |
+
|
| 127 |
+
**Two contexts per person are enough.** Measured on contexts held out for people the matrix had seen: 0.9875 at two, 0.9945 at five, 0.9986 at twenty. Forty is a waste.
|
| 128 |
+
|
| 129 |
+
**Five hundred names generalise to names never seen.** A held-out band at popularity ranks 501 to 540, absent from every calibration, reconstructs at 0.8795 against 0.8844 for the covered ones. Covering 500 people does not teach 500 names; it teaches the map how to handle that region of the space. Going to several thousand would buy very little.
|
| 130 |
+
|
| 131 |
+
**What still fails is not the corpus.** Characters whose identity is a mask rather than a face come out as a stranger wearing the right costume. People whose fame predates the era when everything was photographed come out wrong or generic. And some names fail on the plain 32B too, so run the reference before blaming the projection — that check has overturned three of my own conclusions.
|
| 132 |
+
|
| 133 |
+
## Where the calibration data comes from
|
| 134 |
+
|
| 135 |
+
The general corpus is [GokuScraper/seedance-2-prompts-datasets](https://huggingface.co/datasets/GokuScraper/seedance-2-prompts-datasets), filtered to prompts of fifteen words or more and deduplicated: 8632 lines, median 128 words. The 500 named people come from a TMDB export published on Kaggle, ranked by popularity, with transliterated names dropped beyond rank 1000.
|
| 136 |
+
|
| 137 |
+
Around each name, five short prompts are generated from templates, and two longer ones in MiniMax H3's section format are written by Mistral Small and Gemini Flash Lite, half each. Everything needed to rebuild the corpus is in the node's `calibration/` folder, including the system prompt the long prompts were written from.
|
| 138 |
+
|
| 139 |
+
## The residual network
|
| 140 |
+
|
| 141 |
+
The `-mlp` files carry a `d_in → 16384 → 5120` network with a GELU, added to the matrix rather than replacing it. Its last layer is initialised to zero, so at the first step the model reproduces the matrix exactly and can only improve on it. It is worth 0.05 to 0.08 of cosine, four times what multiplying the corpus by eleven buys the linear map.
|
| 142 |
+
|
| 143 |
+
**Which of the two renders better is not settled.** The cosine does not predict it — that is the single most repeated lesson of this project. Try both on your own prompts.
|
| 144 |
+
|
| 145 |
+
Two things measured while building it. Width beats depth: at equal parameter count, two hidden layers of 8192 reach 0.7691 against 0.7944 for one layer of 16384. And a residual extrapolates worse than a matrix does — outside the corpus it saw, a linear map degrades gracefully while the network collapses.
|
| 146 |
+
|
| 147 |
+
## Measured results
|
| 148 |
+
|
| 149 |
+
| | 4B | 8B |
|
| 150 |
+
|---|---|---|
|
| 151 |
+
| matrix, no names | 0.7169 | 0.7528 |
|
| 152 |
+
| matrix + residual | 0.7944 | 0.7970 |
|
| 153 |
+
| matrix, names covered | 0.7095 | 0.7466 |
|
| 154 |
+
| matrix + residual, names covered | 0.7930 | **0.8037** |
|
| 155 |
+
|
| 156 |
+
A cosine of 0.79 sounds poor and is not — the DiT tolerates far more than the metric suggests. What holds up in actual generation: simple prompts, structured multi-shot prompts with several distinct cuts and no bleed between them, fl2va with first and last frame, ref2va with a reference image, and since 0.1.3 ref2va with a reference video.
|
| 157 |
+
|
| 158 |
+
Fidelity does **not** collapse on short prompts: measured per-token cosine goes from 0.937 at 80 words to 0.908 at 2 words, once the attention sink is handled.
|
| 159 |
+
|
| 160 |
+
## Speech
|
| 161 |
+
|
| 162 |
+
The first release lost non-English speech: a French line came out half Spanish, and the 8B put everything in English. That was the clearest regression and I could not explain it then.
|
| 163 |
+
|
| 164 |
+
With `mmh3-8b-ClipProj-celeb-mlp`, a three-shot clip carrying English, French and Spanish comes out like the 32B does, and the audio level gap measured against the reference has gone from 7.6 dB to 3.5.
|
| 165 |
+
|
| 166 |
+
Part of what was blamed on the projection was not the projection. A line that fills more than about two thirds of its shot comes out slurred whatever encoder produced the conditioning — the fix is a longer shot, not a better matrix. And MiniMax H3 expects speech wrapped in `<d>[Language] ...</d>` with a stable speaker id declared beforehand; without that, one voice with one accent is used for the whole clip. Neither of those is documented here because neither is ours, but both cost me a day.
|
| 167 |
+
|
| 168 |
+
## Run the controls first
|
| 169 |
+
|
| 170 |
+
The two control matrices exist to prove the learned matrix is doing the work rather than the diffusion model. Same prompt, same seed, only the matrix changes:
|
| 171 |
+
|
| 172 |
+
| Matrix | Output for *"a red ball on a wood table"* |
|
| 173 |
+
|---|---|
|
| 174 |
+
| `mmh3-ClipProj-control-zero` | a countryside landscape — the prompt is entirely ignored |
|
| 175 |
+
| `mmh3-ClipProj-control-identity` | a golden object in flames — unusable |
|
| 176 |
+
| a learned matrix | the red ball on a wood table |
|
| 177 |
+
|
| 178 |
+
`‖W_identity‖ = 50.6` against `‖W_learned‖ = 52.4` — near-identical energy, so the difference is structural, not a matter of scale.
|
| 179 |
+
|
| 180 |
+
**If the identity control ever looks fine, the learned matrix adds nothing — and you want to know that before trusting it.**
|
| 181 |
+
|
| 182 |
+
## What is in obsolete/
|
| 183 |
+
|
| 184 |
+
The previous matrices, kept because a comparison posted on r/StableDiffusion ran on them and the links have to keep working. They have no name coverage and are calibrated on a corpus thirty times smaller. There is no reason to prefer them.
|
| 185 |
+
|
| 186 |
+
Among them, the `CONDPROJ` pair, and the story is worth telling because the mistake was instructive.
|
| 187 |
+
|
| 188 |
+
The DiT does not consume the conditioning as it arrives: it first passes it through `condition_proj`, a `Linear(5120 → 5376)` feeding the token refiner. That layer's spectrum is very uneven — a factor of 45 between the top and bottom deciles of its singular values, 52 % of the energy in 10 % of the directions. Plain ridge regression ignores this and spends as much effort on a direction the DiT will multiply by 0.10 as on one it will multiply by 37. Calibrating against the **output** of that layer instead, then mapping back through the pseudo-inverse, should therefore minimise the error the DiT actually sees. The cosine went from 0.697 to 0.845 on the 4B and 0.731 to 0.860 on the 8B.
|
| 189 |
+
|
| 190 |
+
Then I compared what the two matrices actually output:
|
| 191 |
+
|
| 192 |
+
```
|
| 193 |
+
4B CONDPROJ against unweighted, same corpus cosine 0.999998
|
| 194 |
+
8B CONDPROJ against unweighted, same corpus cosine 0.999999
|
| 195 |
+
```
|
| 196 |
+
|
| 197 |
+
They are the same function. Unregularised least squares is invariant to an invertible linear transform of the targets, so fitting in one space and mapping back recovers the same map; only the ridge penalty breaks that invariance, and with 37 851 training tokens against λ = 1000 it barely binds. The entire gain was an artefact of measuring in a different space.
|
| 198 |
+
|
| 199 |
+
*The idea came from u/stddealer on r/StableDiffusion, and it was a good one. The measurement is on me: I published the cosine before checking whether the matrix had changed at all.*
|
| 200 |
+
|
| 201 |
+
## Known limitations
|
| 202 |
+
|
| 203 |
+
**Quantisation costs facts.** Comparing `int8_convrot` against `bf16` on factual recall shows errors appearing under quantisation. Fine for general use, worth knowing if your prompts lean on proper nouns.
|
| 204 |
+
|
| 205 |
+
**Masks defeat identity.** A character recognised by a costume rather than a face comes out as an unknown person in the right suit. No corpus fixes that, because the identity is not in the name's representation to begin with.
|
| 206 |
+
|
| 207 |
+
**Counting is unreliable, and not because of the projection.** Ask for three of something and you get four, on the 32B too. Enumerating works better than announcing a number.
|
| 208 |
+
|
| 209 |
+
## Required models
|
| 210 |
+
|
| 211 |
+
| Role | Model |
|
| 212 |
+
|---|---|
|
| 213 |
+
| Diffusion model + VAEs | [Comfy-Org/MiniMax-H3](https://huggingface.co/Comfy-Org/MiniMax-H3) |
|
| 214 |
+
| Text encoder, 4B | [Comfy-Org/Krea-2](https://huggingface.co/Comfy-Org/Krea-2) → `text_encoders/qwen3vl_4b_fp8_scaled.safetensors` |
|
| 215 |
+
| Text encoder, 8B | any ComfyUI-format Qwen3-VL-8B (the 8B matrices expect 4096 input dims) |
|
| 216 |
+
|
| 217 |
+
The 32B text encoder is **no longer needed** — that is the entire point.
|
| 218 |
+
|
| 219 |
+
## Licence and responsibility
|
| 220 |
+
|
| 221 |
+
These matrices are released under **MIT**, like the node.
|
| 222 |
+
|
| 223 |
+
They are derived from the activations of both models, and their legal status is unclear. They are provided as-is, for research, with no claim of ownership over anything derived from the underlying models.
|
| 224 |
+
|
| 225 |
+
- **Qwen3-VL** is published by Alibaba under **Apache 2.0**. Read and comply with its terms and acceptable-use policy.
|
| 226 |
+
- **MiniMax H3** ships under a **custom licence**. Read it before any use, particularly commercial.
|
| 227 |
+
|
| 228 |
+
This project is **not affiliated with, endorsed by, or connected to** Alibaba / Qwen, MiniMax, or Comfy Org.
|
| 229 |
+
|
| 230 |
+
You remain responsible for what you generate and for complying with the licences of every model you load.
|
| 231 |
+
|
| 232 |
+
## Credits
|
| 233 |
+
|
| 234 |
+
Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number quoted was measured on real hardware, not estimated: where a prediction turned out wrong, the measurement won and the text was corrected. Three claims in the previous version of this file were wrong and are corrected here.
|
RELEASE_NOTES_v3.md
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# ClipProj v3
|
| 2 |
+
|
| 3 |
+
Four new matrices, calibrated against `qwen3vl_32b_minimax_h3_nvfp4_awq` — the
|
| 4 |
+
base MiniMax H3 encoder, the one a plain `Load CLIP` gives you.
|
| 5 |
+
|
| 6 |
+
Requires node **0.1.13** or later. The `-v3-mlp` files carry no linear matrix,
|
| 7 |
+
and an older node raises `KeyError: 'W'` on load.
|
| 8 |
+
|
| 9 |
+
## What changed
|
| 10 |
+
|
| 11 |
+
**Body descriptions.** While testing v2 we found that naming one part of a
|
| 12 |
+
body could rewrite the whole of it — build, height and face shifting together,
|
| 13 |
+
none of it asked for. Those matrices were calibrated against a modified 32B.
|
| 14 |
+
v3 is calibrated against the stock encoder, and we no longer observe the
|
| 15 |
+
problem: a build and an attribute stated in the same prompt are honoured
|
| 16 |
+
independently.
|
| 17 |
+
|
| 18 |
+
**Everything is measurably closer to the 32B.** Measured on a single reference
|
| 19 |
+
prompt, every projection encoded against the same stock 32B, so the numbers are
|
| 20 |
+
comparable to each other — which the figures published with v1 and v2 were not,
|
| 21 |
+
having been measured against different targets and in different spaces. The
|
| 22 |
+
`cos_test` written inside each file is the training-time figure against that
|
| 23 |
+
campaign's own target and does **not** compare across versions; the comparable
|
| 24 |
+
number is stored separately as `cos_prompt_reference`.
|
| 25 |
+
|
| 26 |
+
The render pipeline was verified deterministic before measuring anything: the
|
| 27 |
+
same workflow run twice, thirteen days apart, produced two files with identical
|
| 28 |
+
SHA-256. Any difference below therefore comes from the projection and nothing
|
| 29 |
+
else.
|
| 30 |
+
|
| 31 |
+
| projection | mean cosine vs 32B |
|
| 32 |
+
|---|---|
|
| 33 |
+
| 8B v3-mlp | 0.9449 |
|
| 34 |
+
| 8B v2 `celeb-mlp` | 0.9393 |
|
| 35 |
+
| 4B v3-mlp | 0.9381 |
|
| 36 |
+
| 4B v2 `celeb-mlp` | 0.9293 |
|
| 37 |
+
| 8B v3 | 0.9289 |
|
| 38 |
+
| 4B v3 | 0.9193 |
|
| 39 |
+
|
| 40 |
+
**The 8B now sees image tokens.** Until v3 the image corpus had only ever been
|
| 41 |
+
encoded with a 4B student, so both 8B matrices projected vision tokens without
|
| 42 |
+
having seen a single one — while the node accepts a reference image. Measured on
|
| 43 |
+
100 held-out images, on the raw conditioning the diffusion model actually
|
| 44 |
+
receives:
|
| 45 |
+
|
| 46 |
+
| | vision tokens | text in the same sequences |
|
| 47 |
+
|---|---|---|
|
| 48 |
+
| 8B residual, before | 0.7692 | 0.9085 |
|
| 49 |
+
| 8B residual, after | 0.8578 | 0.9605 |
|
| 50 |
+
| 8B matrix, before | 0.7845 | 0.8926 |
|
| 51 |
+
| 8B matrix, after | 0.8457 | 0.9361 |
|
| 52 |
+
|
| 53 |
+
That costs 0.0027 of pure-text cosine on the residual and 0.0013 on the matrix —
|
| 54 |
+
which is why the 8B figures above are slightly below what a text-only corpus
|
| 55 |
+
would have given.
|
| 56 |
+
|
| 57 |
+
**Prefer the `-mlp` files on the measurement, not on this scene.** They sit
|
| 58 |
+
closer to the 32B, 0.9449 against 0.9289 on the 8B. But on this prompt the five
|
| 59 |
+
renders are faithful, plain matrices included: the pose, the dress, the white
|
| 60 |
+
pieces, the cat, the straw hat, the laundry, the bouncing knee all hold on all
|
| 61 |
+
five. A tightly written prompt survives even the linear baseline, and the
|
| 62 |
+
difference between the files shows up in the numbers well before it shows up on
|
| 63 |
+
screen.
|
| 64 |
+
|
| 65 |
+
## What this does not fix, and will not
|
| 66 |
+
|
| 67 |
+
A projection cannot invent information the small encoder never had. Five hours
|
| 68 |
+
of training on a 3090, two more to encode the dataset, six and a half million
|
| 69 |
+
tokens — none of that changes what a 4B model wrote down in the first place.
|
| 70 |
+
The result is an approximation of the 32B's conditioning, not a copy of it.
|
| 71 |
+
|
| 72 |
+
In practice the line falls here: **what the prompt states, the projection
|
| 73 |
+
carries; what the prompt leaves open, the model fills from its own prior.**
|
| 74 |
+
Constrain a scene tightly and the projected renders track the 32B closely.
|
| 75 |
+
Leave the set dressing unstated — a cat somewhere, laundry on a line, furniture
|
| 76 |
+
— and it will be furnished differently. That is not a defect to be tuned away,
|
| 77 |
+
it is what a 20-degree angle between two conditioning vectors looks like on
|
| 78 |
+
screen.
|
| 79 |
+
|
| 80 |
+
## Training data
|
| 81 |
+
|
| 82 |
+
The aim was to activate as much of the encoder's weight space as possible
|
| 83 |
+
rather than to cover one domain deeply. A matrix only learns to project the
|
| 84 |
+
directions it has actually seen used, so the corpus deliberately mixes
|
| 85 |
+
registers, languages and lengths.
|
| 86 |
+
|
| 87 |
+
| source | tokens |
|
| 88 |
+
|---|---|
|
| 89 |
+
| cinematic video prompts | 1 342 987 |
|
| 90 |
+
| native H3 format, 4 length draws | 3 169 879 |
|
| 91 |
+
| explicit register | 544 073 |
|
| 92 |
+
| Chinese | 532 302 |
|
| 93 |
+
| celebrity prompts, long form | 314 516 |
|
| 94 |
+
| filler sequences | 149 917 |
|
| 95 |
+
| celebrity prompts, short form | 99 668 |
|
| 96 |
+
| images, 3 blocks, 1 700 images | 349 244 |
|
| 97 |
+
| **total, both students** | **6 502 586** |
|
| 98 |
+
|
| 99 |
+
Both students now see the same corpus, images included — which makes the 4B and
|
| 100 |
+
the 8B comparable to each other for the first time.
|
| 101 |
+
|
| 102 |
+
3 331 prompts for fitting, one in fifty held out for measurement. Tap 24 on
|
| 103 |
+
both students. Sequence lengths are drawn at random rather than truncated to a
|
| 104 |
+
fixed size, so a given word appears at many different positions instead of
|
| 105 |
+
always the same ones.
|
| 106 |
+
|
| 107 |
+
## The demo folder
|
| 108 |
+
|
| 109 |
+
`demo/` holds five renders of the same scene. Same prompt, same seed 42, same
|
| 110 |
+
8 steps, same turbo LoRA, same DiT, same VAE — only the projection changes. The
|
| 111 |
+
pipeline is reproducible bit for bit, verified by running it twice and comparing
|
| 112 |
+
the decoded video and audio streams, so every difference between these five
|
| 113 |
+
comes from the projection and nothing else.
|
| 114 |
+
|
| 115 |
+
| file | conditioning |
|
| 116 |
+
|---|---|
|
| 117 |
+
| `chess-32b-reference.mp4` | Qwen3-VL-32B, 15.7 GB |
|
| 118 |
+
| `chess-8b-mlp.mp4` | Qwen3-VL-8B + `v3-mlp`, 10.0 GB |
|
| 119 |
+
| `chess-4b-mlp.mp4` | Qwen3-VL-4B + `v3-mlp`, 4.8 GB |
|
| 120 |
+
| `chess-8b-ridge.mp4` | Qwen3-VL-8B + `v3`, matrix only |
|
| 121 |
+
| `chess-4b-ridge.mp4` | Qwen3-VL-4B + `v3`, matrix only |
|
| 122 |
+
| `chess-comparison.mp4` | the five in sequence, labelled |
|
| 123 |
+
| `chess-prompt.txt` | the prompt, verbatim |
|
| 124 |
+
|
| 125 |
+
The comparison runs matrix-only first, then the 32B, then the residuals — so the
|
| 126 |
+
reference sits in the middle and each half is read against it.
|
| 127 |
+
|
| 128 |
+
**You will not reproduce these files byte for byte, and that is normal.** Noticed
|
| 129 |
+
while testing something else, so take it as an observation rather than a study:
|
| 130 |
+
the result depends on the model of GPU the encoder runs on. Four cards, one
|
| 131 |
+
prompt, one seed, four different outputs — while two different RTX 3090s gave
|
| 132 |
+
byte-identical video. Not the architecture either: the 3060 and the 3090 are
|
| 133 |
+
both Ampere and disagree. Encoding the same prompt on two cards gives
|
| 134 |
+
conditioning that matches to a relative error of 7 × 10⁻⁷; eight denoising steps
|
| 135 |
+
turn that into a different piece of furniture. On one machine everything here is
|
| 136 |
+
reproducible to the bit, which is what makes the five-way comparison meaningful.
|
| 137 |
+
|
| 138 |
+
Watch her knee. The prompt asks for it three times, ending on a sentence of its
|
| 139 |
+
own: *"Her knee never stops bouncing."* It is the most redundant instruction in
|
| 140 |
+
the text, it is a continuous involuntary motion with no narrative purpose, and
|
| 141 |
+
it is the clearest single indicator that a projection carried what was written.
|
| 142 |
+
Then watch the cat, the laundry and the furniture, which are named once and
|
| 143 |
+
anchored nowhere — those move, and they are supposed to.
|
| 144 |
+
|
| 145 |
+
One thing none of the five gets right, including the 32B: she lifts a knight and
|
| 146 |
+
does not put it back on the same square. Object permanence through an occluding
|
| 147 |
+
hand on a grid of sixty-four identical squares is a limit of the video model, not
|
| 148 |
+
of the conditioning. It is listed here so nobody attributes it to the projection.
|
| 149 |
+
|
| 150 |
+
## Files
|
| 151 |
+
|
| 152 |
+
| file | size | structure | encoder |
|
| 153 |
+
|---|---|---|---|
|
| 154 |
+
| `mmh3-4b-ClipProj-v3-mlp.safetensors` | 503 MB | residual only | Qwen3-VL-4B |
|
| 155 |
+
| `mmh3-8b-ClipProj-v3-mlp.safetensors` | 604 MB | residual only | Qwen3-VL-8B |
|
| 156 |
+
| `mmh3-4b-ClipProj-v3.safetensors` | 26 MB | matrix only | Qwen3-VL-4B |
|
| 157 |
+
| `mmh3-8b-ClipProj-v3.safetensors` | 42 MB | matrix only | Qwen3-VL-8B |
|
| 158 |
+
|
| 159 |
+
The residuals use a hidden width of 32 768 against 16 384 in v2, which is where
|
| 160 |
+
the extra download size comes from.
|
bench3.1/README.md
ADDED
|
@@ -0,0 +1,506 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: mit
|
| 3 |
+
tags:
|
| 4 |
+
- comfyui
|
| 5 |
+
- minimax-h3
|
| 6 |
+
- text-to-video
|
| 7 |
+
- qwen3-vl
|
| 8 |
+
- text-encoder
|
| 9 |
+
- multilingual
|
| 10 |
+
base_model:
|
| 11 |
+
- Comfy-Org/MiniMax-H3
|
| 12 |
+
- Qwen/Qwen3-VL-4B-Instruct
|
| 13 |
+
- Qwen/Qwen3-VL-8B-Instruct
|
| 14 |
+
library_name: comfyui
|
| 15 |
+
---
|
| 16 |
+
|
| 17 |
+
# ClipProj v3.1 — measured against the 32B's own variance
|
| 18 |
+
|
| 19 |
+
**Four projection matrices that let a Qwen3-VL-4B or 8B replace the Qwen3-VL-32B text encoder of MiniMax H3.**
|
| 20 |
+
|
| 21 |
+
**15.0 GB → 4.6 GB**, with no change to the diffusion model, the VAEs or the sampler.
|
| 22 |
+
|
| 23 |
+
This release is not about a new architecture. It is about finally knowing **how good these things are**, because the previous numbers could not tell me. This card is mostly the measurement, and the measurement changed three of my own conclusions.
|
| 24 |
+
|
| 25 |
+
Requires the custom node: **[github.com/nicolab28/ComfyUI-ClipProj](https://github.com/nicolab28/ComfyUI-ClipProj)**
|
| 26 |
+
|
| 27 |
+
---
|
| 28 |
+
|
| 29 |
+
## Read this before the tables: these are metrics, not verdicts
|
| 30 |
+
|
| 31 |
+
**I do not speak these eleven languages.** I cannot tell you whether a render sounds right, and I have not
|
| 32 |
+
asked anyone who can. No native speaker has listened to any of the 297 speech renders on this page.
|
| 33 |
+
|
| 34 |
+
So nothing below is a judgement of quality. Every figure is a **distance between two automatic
|
| 35 |
+
transcriptions** — what one machine wrote down from the reference, against what it wrote down from the
|
| 36 |
+
projection. That is all it is, and it is worth being explicit about what that does and does not capture:
|
| 37 |
+
|
| 38 |
+
**What the numbers do capture.** Whether the same words and the same sounds come out. Two instruments are
|
| 39 |
+
used precisely because each is wrong in a known direction: **Whisper** has a language model inside and
|
| 40 |
+
corrects a slurred word into the most probable real one, so it *under*-reports pronunciation defects — a
|
| 41 |
+
lower bound. **ZIPA** has no lexical decoder at all and counts every shift in realisation as an error, so
|
| 42 |
+
it *over*-reports — an upper bound. What a listener would notice lies between them, and neither number
|
| 43 |
+
alone is the answer.
|
| 44 |
+
|
| 45 |
+
**What they do not capture.** Prosody, rhythm, timbre, naturalness — everything that makes speech sound
|
| 46 |
+
native rather than merely correct. A render scoring 98.8 here could still sound foreign to someone who
|
| 47 |
+
speaks the language. These metrics cannot see that, and neither can I.
|
| 48 |
+
|
| 49 |
+
**So read a score as "close to the 32B, according to this instrument"** — never as "good". If you speak
|
| 50 |
+
one of these languages, your ear outranks every table below, and I would genuinely like to hear what it
|
| 51 |
+
tells you.
|
| 52 |
+
|
| 53 |
+
---
|
| 54 |
+
|
| 55 |
+
## What changed in v3.1: giving every script its share
|
| 56 |
+
|
| 57 |
+
v3 was calibrated on a corpus that was overwhelmingly English, with the other languages bolted on
|
| 58 |
+
afterwards as a top-up. v3.1 **adds text and tagged prompts until every writing system carries roughly
|
| 59 |
+
comparable weight** — English excepted, because the prompt format itself is English: the sections, the
|
| 60 |
+
tags and the descriptions are all written in it, so it stays the majority no matter what.
|
| 61 |
+
|
| 62 |
+
Measured share of the v3.1 corpus:
|
| 63 |
+
|
| 64 |
+
| Script | Languages | Tagged prompts | Raw text | Share |
|
| 65 |
+
|---|---|---|---|---|
|
| 66 |
+
| Latin — base | English: the original corpus, image lots and register lots | *base* | — | **68.3 %** |
|
| 67 |
+
| Han | zh | ✓ | ✓ | 6.7 % |
|
| 68 |
+
| Hangul | ko | ✓ | ✓ | 6.6 % |
|
| 69 |
+
| Latin, accented | fr | ✓ | ✓ | 6.3 % |
|
| 70 |
+
| Arabic | ar | ✓ | ✓ **(new)** | 4.1 % |
|
| 71 |
+
| Latin | es, de, it, pt | ✓ | — | 1.3 % each |
|
| 72 |
+
| Cyrillic | ru | ✓ | — | 1.3 % |
|
| 73 |
+
|
| 74 |
+
The rule behind those numbers: **a script that inherits nothing from Latin needs raw text**; a Latin
|
| 75 |
+
script only needs tagged prompts, because the alphabet is already covered and roughly 250 tags are
|
| 76 |
+
enough to attach a language to it. That is why Chinese, Korean and French carry raw lots and Spanish
|
| 77 |
+
does not.
|
| 78 |
+
|
| 79 |
+
**The one genuinely new lot is raw Arabic** — 550 000 characters. Arabic was the last non-Latin script
|
| 80 |
+
still living on tagged prompts alone.
|
| 81 |
+
|
| 82 |
+
The training itself was also restarted from scratch rather than topped up. A network keeps the order it
|
| 83 |
+
learned in: whatever comes last weighs more, and lowering the learning rate on a top-up run does not
|
| 84 |
+
remove that imbalance, it only arbitrates between preserving what was acquired and correcting it. Same
|
| 85 |
+
architecture and same hyper-parameters as v3 — `hidden 32768`, `depth 1`, `tap 24`, `lr 1e-3`, no linear
|
| 86 |
+
path — so what the benchmark below compares is the corpus, not the recipe.
|
| 87 |
+
|
| 88 |
+
### What it buys
|
| 89 |
+
|
| 90 |
+
Phoneme errors against the 32B, averaged over the four files of each generation, the three seeds and
|
| 91 |
+
compared against the threshold:
|
| 92 |
+
|
| 93 |
+
| | threshold | v3 | **v3.1** | |
|
| 94 |
+
|---|---|---|---|---|
|
| 95 |
+
| es | 0.0 | 1.9 | **0.5** | −74 % |
|
| 96 |
+
| de | 2.7 | 9.8 | **3.6** | −64 % |
|
| 97 |
+
| fr | 4.0 | 8.2 | **3.0** | −64 % |
|
| 98 |
+
| it | 2.3 | 4.2 | **1.6** | −63 % |
|
| 99 |
+
| ru | 10.7 | 21.5 | **13.6** | −37 % |
|
| 100 |
+
| ar | 6.3 | 11.1 | **7.6** | −32 % |
|
| 101 |
+
| zh | 3.3 | 3.0 | **2.2** | −25 % |
|
| 102 |
+
| ja | 6.7 | 7.5 | **6.4** | −14 % |
|
| 103 |
+
| ko | 11.3 | 12.8 | **11.5** | −10 % |
|
| 104 |
+
| pt | 17.0 | 25.1 | **24.1** | −4 % |
|
| 105 |
+
| en | 0.0 | **0.2** | 0.5 | +0.3 |
|
| 106 |
+
| **mean** | 5.8 | **9.6** | **6.8** | **−29 %** |
|
| 107 |
+
|
| 108 |
+
The European languages, which v3 only ever saw as a top-up, gain 60 to 74 %. Arabic gains 32 %, which is
|
| 109 |
+
where the new raw-text lot shows up. Russian gains 37 %.
|
| 110 |
+
|
| 111 |
+
**Raw text does not predict the outcome.** Rapported to each language's own threshold, the two groups
|
| 112 |
+
overlap completely:
|
| 113 |
+
|
| 114 |
+
| | script | raw text | threshold | v3 | v3.1 | × threshold |
|
| 115 |
+
|---|---|---|---|---|---|---|
|
| 116 |
+
| zh | Han | ✓ | 3.3 | 3.0 | 2.2 | **0.67** |
|
| 117 |
+
| it | Latin | — | 2.3 | 4.2 | 1.6 | **0.68** |
|
| 118 |
+
| fr | Latin | ✓ | 4.0 | 8.2 | 3.0 | 0.75 |
|
| 119 |
+
| ja | Kana/Kanji | — | 6.7 | 7.5 | 6.4 | 0.96 |
|
| 120 |
+
| ko | Hangul | ✓ | 11.3 | 12.8 | 11.5 | 1.01 |
|
| 121 |
+
| ar | Arabic | ✓ | 6.3 | 11.1 | 7.6 | 1.20 |
|
| 122 |
+
| ru | Cyrillic | — | 10.7 | 21.5 | 13.6 | 1.27 |
|
| 123 |
+
| de | Latin | — | 2.7 | 9.8 | 3.6 | 1.34 |
|
| 124 |
+
| pt | Latin | — | 17.0 | 25.1 | 24.1 | 1.42 |
|
| 125 |
+
|
| 126 |
+
Languages with a raw lot run 0.67 to 1.20; languages without run 0.68 to 1.42. Italian, with tagged
|
| 127 |
+
prompts only, lands second best overall.
|
| 128 |
+
|
| 129 |
+
**Russian is the clearest case.** It is the only non-Latin script here with no raw text and nothing to
|
| 130 |
+
inherit — Japanese borrows kanji from the Chinese lots, Latin scripts borrow the alphabet from English —
|
| 131 |
+
and it still gains **37 %** between v3 and v3.1 on tagged prompts alone, finishing ahead of German and
|
| 132 |
+
Portuguese, which are Latin. The rule written in the build scripts — *250 tags are enough once the
|
| 133 |
+
alphabet is covered* — evidently extends to Cyrillic, which the Qwen3-VL tokenizer covers natively.
|
| 134 |
+
|
| 135 |
+
So raw text is what an **unseen script** needs, not what a language needs. Where a script is already in
|
| 136 |
+
the tokenizer's reach, tags carry it.
|
| 137 |
+
|
| 138 |
+
**English pays for it, and the bill is half a phoneme out of 77.** That is the whole cost of rebalancing:
|
| 139 |
+
v3 was 0.2 errors, v3.1 is 0.5, both far below anything audible and below what a single seed resolves.
|
| 140 |
+
Portuguese barely moves, but nothing moves in Portuguese — the reference itself scatters by 17 there.
|
| 141 |
+
|
| 142 |
+
The net effect is a change of category rather than a better score. v3 sits at **1.46 to 1.77 times** the
|
| 143 |
+
threshold; v3.1 sits at **1.09 to 1.20**. From measurably worse than a seed change, to indistinguishable
|
| 144 |
+
from one.
|
| 145 |
+
|
| 146 |
+
---
|
| 147 |
+
|
| 148 |
+
## The problem with every number I published before
|
| 149 |
+
|
| 150 |
+
A cosine of 0.79, or "23 character errors out of 869" — neither has a scale. Is 23 good? Compared to what? Zero errors is not the right target either, because **the 32B does not reproduce itself**. Change nothing but the seed and it re-pronounces the sentence differently.
|
| 151 |
+
|
| 152 |
+
So the reference is not perfection. It is the 32B compared to itself, same prompt, different seed:
|
| 153 |
+
|
| 154 |
+
| | 32B against itself |
|
| 155 |
+
|---|---|
|
| 156 |
+
| Speech | **5.8 phonemes out of 75** (7.8 %) |
|
| 157 |
+
| Image | **0.9552** SigLIP2 cosine (floor: 0.5313) |
|
| 158 |
+
|
| 159 |
+
That gap is the unit. Everything below is normalised so that **32B = 100**:
|
| 160 |
+
|
| 161 |
+
- **100** — swapping the encoder moves the output as much as changing the seed
|
| 162 |
+
- **above 100** — it moves it less
|
| 163 |
+
- **below 100** — it moves it more
|
| 164 |
+
|
| 165 |
+
Below that threshold you are no longer measuring the projection. You are measuring the generator.
|
| 166 |
+
|
| 167 |
+
---
|
| 168 |
+
|
| 169 |
+
## Files
|
| 170 |
+
|
| 171 |
+
Put them in `ComfyUI/models/clip_projections/`.
|
| 172 |
+
|
| 173 |
+
| File | Encoder | Head | Size | Encoder + projection |
|
| 174 |
+
|---|---|---|---|---|
|
| 175 |
+
| `mmh3-4b-ClipProj-v3.1` | any Qwen3-VL-4B | ridge | 26 MB | **4.6 GB** |
|
| 176 |
+
| `mmh3-4b-ClipProj-v3.1-mlp` | any Qwen3-VL-4B | ridge + residual | 481 MB | 5.1 GB |
|
| 177 |
+
| `mmh3-8b-ClipProj-v3.1` | any Qwen3-VL-8B | ridge | 41 MB | 9.6 GB |
|
| 178 |
+
| `mmh3-8b-ClipProj-v3.1-mlp` | any Qwen3-VL-8B | ridge + residual | 577 MB | 10.1 GB |
|
| 179 |
+
|
| 180 |
+
The 8B matrices expect 4096 input dimensions instead of 2560; the node checks the width and refuses a mismatch.
|
| 181 |
+
|
| 182 |
+
---
|
| 183 |
+
|
| 184 |
+
## The benchmark
|
| 185 |
+
|
| 186 |
+
Nothing here is a single render. Every figure comes from **three seeds — 42, 100 000 and 100 000 000** — chosen far apart so no one can suspect they are correlated.
|
| 187 |
+
|
| 188 |
+
| | volume |
|
| 189 |
+
|---|---|
|
| 190 |
+
| Speech | 297 renders — 9 conditionings × 11 languages × 3 seeds |
|
| 191 |
+
| Image | 405 renders — 9 conditionings × 15 prompts × 3 seeds |
|
| 192 |
+
|
| 193 |
+
**Speech** is scored in phonemes, by [ZIPA-CR-large](https://huggingface.co/anyspeech/zipa-large-crctc-ns-800k) (88 languages, no lexical decoder — it will not silently repair a botched syllable into a real word), against the 32B **of the same seed**. Distances are Levenshtein throughout.
|
| 194 |
+
|
| 195 |
+
**Image** is scored by SigLIP2 so400m, both against the 32B's render and against the prompt itself — on that second axis the 32B is just one column among nine.
|
| 196 |
+
|
| 197 |
+
Languages: en, fr, es, de, it, pt, ru, ar, zh, ja, ko.
|
| 198 |
+
|
| 199 |
+
---
|
| 200 |
+
|
| 201 |
+
## Results
|
| 202 |
+
|
| 203 |
+
| Conditioning | Speech | ± | Image | Prompt |
|
| 204 |
+
|---|---|---|---|---|
|
| 205 |
+
| **32B** *(reference)* | **100.0** | — | **100.0** | **100.0** |
|
| 206 |
+
| `8b-ClipProj-v3.1` | **98.8** | ±1.4 | 100.0 | 100.4 |
|
| 207 |
+
| `8b-ClipProj-v3.1-mlp` | 98.2 | ±1.2 | 101.4 | **102.3** |
|
| 208 |
+
| `4b-ClipProj-v3.1` | 97.9 | ±1.0 | 100.1 | 99.4 |
|
| 209 |
+
| `4b-ClipProj-v3.1-mlp` | 97.8 | ±1.1 | 100.9 | 100.3 |
|
| 210 |
+
| *v3 ridge / mlp, 4B and 8B* | *93.2 – 95.6* | | *99.4 – 102.4* | *99.0 – 100.7* |
|
| 211 |
+
|
| 212 |
+
`±` is the spread of the score across the three seeds. **Two models separated by less than that are not separated at all.**
|
| 213 |
+
|
| 214 |
+
### The raw counts behind the speech score
|
| 215 |
+
|
| 216 |
+
Three metrics, three units, never added together. **PER** counts phonemes over three seeds on the current
|
| 217 |
+
protocol; **WER** and **CER** count words and characters as Whisper hears them, single-seed on the earlier
|
| 218 |
+
0.3 MP protocol. They are listed side by side because they disagree in useful ways — see the language
|
| 219 |
+
breakdown below.
|
| 220 |
+
|
| 221 |
+
| | **PER** (3 seeds) | | **WER** (1 seed) | **CER** (1 seed) |
|
| 222 |
+
|---|---|---|---|---|
|
| 223 |
+
| **32B** *(reference)* | **0 / 2469** | — | 6 / 174 | 6 / 869 |
|
| 224 |
+
| *32B against itself* | *~193 / 2469* | *7.8 %* | — | — |
|
| 225 |
+
| `8b-v3.1` | **211 / 2469** | 8.5 % | **10 / 174** | **14 / 869** |
|
| 226 |
+
| `8b-v3.1-mlp` | 222 / 2469 | 9.0 % | 15 / 174 | 28 / 869 |
|
| 227 |
+
| `4b-v3.1-mlp` | 230 / 2469 | 9.3 % | 13 / 174 | 23 / 869 |
|
| 228 |
+
| `4b-v3.1` | 232 / 2469 | 9.4 % | 15 / 174 | 30 / 869 |
|
| 229 |
+
| `4b-v3-mlp` | 281 / 2469 | 11.4 % | 18 / 174 | 32 / 869 |
|
| 230 |
+
| `8b-v3-mlp` | 317 / 2469 | 12.8 % | 17 / 174 | 29 / 869 |
|
| 231 |
+
| `8b-v3` | 326 / 2469 | 13.2 % | 23 / 174 | 46 / 869 |
|
| 232 |
+
| `4b-v3` | 341 / 2469 | 13.8 % | 19 / 174 | 36 / 869 |
|
| 233 |
+
|
| 234 |
+
The 32B scores 0 on PER by construction — it *is* the reference. The row below it is the meaningful one:
|
| 235 |
+
compared to **itself** on another seed it drifts by about 7.8 %, and the four v3.1 files sit at 8.5 to
|
| 236 |
+
9.4 %. The v3 files sit at 11.4 to 13.8 %, clear of that band.
|
| 237 |
+
|
| 238 |
+
WER and CER rank the files in nearly the same order, which is the point of quoting both: `8b-v3.1` leads
|
| 239 |
+
all three metrics, and no v3 file beats any v3.1 file on any of them.
|
| 240 |
+
|
| 241 |
+
### Why some scores exceed 100 — and why that is not "better than the 32B"
|
| 242 |
+
|
| 243 |
+
The two image columns do not share a reference, and neither exceedance means what it looks like.
|
| 244 |
+
|
| 245 |
+
**Prompt.** This axis is `cos(image embedding, prompt embedding)`. The 32B is **not** the reference here —
|
| 246 |
+
it is one column among nine, and its value is set to 100 only to give the scale a fixed point. Nothing
|
| 247 |
+
requires it to be the best, and it demonstrably is not: on the prompt asking for a loaf **cut in two**, it
|
| 248 |
+
renders a single piece on two seeds out of three. A projection that follows the description more closely
|
| 249 |
+
earns a higher cosine, legitimately.
|
| 250 |
+
|
| 251 |
+
**Image.** Here 100 *is* the 32B against itself, but the comparison is asymmetric: the threshold pits
|
| 252 |
+
`32B(seed 42)` against `32B(seed 7391)` — two different draws — while a projection is compared to
|
| 253 |
+
`32B(seed 42)`, the **same** draw. It plays with its reference's seed, so the draw noise is removed on its
|
| 254 |
+
side. A score of 101.4 says only *closer to that 32B render than two 32B renders are to each other*.
|
| 255 |
+
|
| 256 |
+
**And none of it is significant.** The nine models span 0.0048 of cosine on the prompt axis, against a
|
| 257 |
+
within-model standard deviation of 0.024 to 0.029 — five times larger. The paired test over 45 cases calls
|
| 258 |
+
all eight projections indistinguishable from the 32B, including the one at 102.3. The +2.3 % is real as a
|
| 259 |
+
measurement and void as a result.
|
| 260 |
+
|
| 261 |
+
### What actually separates
|
| 262 |
+
|
| 263 |
+
**The corpus, not the size and not the head.** All four v3.1 land within one point of each other — 97.8 to 98.8 — while ranging from 4.6 to 10.1 GB. All four v3 sit a clear notch below, 93.2 to 95.6, at identical sizes. A 4B v3.1 beats an 8B v3 by four points while weighing half as much.
|
| 264 |
+
|
| 265 |
+
**Nothing separates in image.** All nine conditionings, v3 included, are at or above the threshold: 99.4 to 102.4. Swapping the 32B for a 4B changes the picture **less than changing the seed does**. On this axis the 32B is not a ceiling — `8b-v3.1-mlp` scores 102.3 for prompt fidelity, and on one prompt asking for a loaf cut in two, the 32B rendered a single piece on two seeds out of three while the 8B ridge rendered two on all three.
|
| 266 |
+
|
| 267 |
+
**4B against 8B does not separate on general pronunciation.** 97.8 against 98.8, for a seed-to-seed spread of ±1.0 to ±1.4. If you need one number: they are the same.
|
| 268 |
+
|
| 269 |
+
**Ridge against MLP does not separate either.** The residual buys nothing measurable in speech. It shows up in image prompt fidelity — 102.3 against 100.4 on the 8B — but that axis has its own noise and I would not choose a file on it.
|
| 270 |
+
|
| 271 |
+
### The image measurements in full
|
| 272 |
+
|
| 273 |
+
Two independent SigLIP2 so400m readings over the same 405 renders. First, resemblance to the 32B's own render, same prompt and same seed — 45 cases per model:
|
| 274 |
+
|
| 275 |
+
| | cosine | std. dev. | % of threshold | worst prompt |
|
| 276 |
+
|---|---|---|---|---|
|
| 277 |
+
| **32B against itself** | **0.9552** | — | **100.0** | — |
|
| 278 |
+
| `8b-v3-mlp` | 0.9653 | 0.0328 | 102.4 | 0.8804 |
|
| 279 |
+
| `8b-v3.1-mlp` | 0.9612 | 0.0363 | 101.4 | 0.8808 |
|
| 280 |
+
| `8b-v3` | 0.9594 | 0.0350 | 101.0 | 0.8910 |
|
| 281 |
+
| `4b-v3.1-mlp` | 0.9590 | 0.0321 | 100.9 | 0.9038 |
|
| 282 |
+
| `4b-v3-mlp` | 0.9588 | 0.0408 | 100.8 | 0.8989 |
|
| 283 |
+
| `4b-v3.1` | 0.9557 | 0.0445 | 100.1 | 0.8766 |
|
| 284 |
+
| `8b-v3.1` | 0.9552 | 0.0448 | 100.0 | 0.8893 |
|
| 285 |
+
| `4b-v3` | 0.9528 | 0.0423 | 99.4 | 0.8861 |
|
| 286 |
+
|
| 287 |
+
*Floor: 0.5313 — two 32B renders sharing no content at all still score that, on style and generator artefacts alone.*
|
| 288 |
+
|
| 289 |
+
**The standard deviation settles it.** It runs 0.032 to 0.045, while the entire spread from best to worst model is 0.0125. The scatter within one model is three to four times the gap between models. Nothing here is a ranking.
|
| 290 |
+
|
| 291 |
+
Second, fidelity to the written prompt — an axis where the 32B is one column among nine rather than the reference:
|
| 292 |
+
|
| 293 |
+
| | cosine | std. dev. | base 100 |
|
| 294 |
+
|---|---|---|---|
|
| 295 |
+
| `8b-v3.1-mlp` | 0.1504 | 0.0252 | **102.3** |
|
| 296 |
+
| `4b-v3-mlp` | 0.1481 | 0.0275 | 100.7 |
|
| 297 |
+
| `8b-v3.1` | 0.1477 | 0.0290 | 100.4 |
|
| 298 |
+
| `4b-v3.1-mlp` | 0.1475 | 0.0249 | 100.3 |
|
| 299 |
+
| **32B** | 0.1471 | 0.0245 | **100.0** |
|
| 300 |
+
| `4b-v3.1` | 0.1462 | 0.0272 | 99.4 |
|
| 301 |
+
| `8b-v3` | 0.1460 | 0.0267 | 99.3 |
|
| 302 |
+
| `8b-v3-mlp` | 0.1458 | 0.0236 | 99.2 |
|
| 303 |
+
| `4b-v3` | 0.1456 | 0.0268 | 99.0 |
|
| 304 |
+
|
| 305 |
+
*Floor: −0.0293 — one scene's image against another scene's prompt.*
|
| 306 |
+
|
| 307 |
+
Same verdict, and harder: the spread across all nine models is 0.0048 for a standard deviation of 0.024 to 0.029, **five times larger**. Four models sit above the 32B and four below, in an order that carries no information.
|
| 308 |
+
|
| 309 |
+
The two image axes do not even agree with each other: `8b-v3-mlp` tops the resemblance table and sits second from last on prompt fidelity. Imitating the 32B and following the prompt are not the same objective — the 32B itself misses prompts.
|
| 310 |
+
|
| 311 |
+
---
|
| 312 |
+
|
| 313 |
+
## What counts as an error
|
| 314 |
+
|
| 315 |
+
One error is **one phoneme inserted, deleted or substituted** relative to what the 32B pronounced — same prompt, same seed. Levenshtein distance, nothing weighted, nothing forgiven.
|
| 316 |
+
|
| 317 |
+
There is no dictionary in the loop. ZIPA transcribes sound to IPA and has no lexical decoder, so it will not quietly repair a botched syllable into a real word the way a speech-to-text engine would. What it writes down is what came out of the speaker.
|
| 318 |
+
|
| 319 |
+
Concretely, on the French line *"la lumière de Marseille"*, seed 42 — the phonemes following `d ɛ` ("de"):
|
| 320 |
+
|
| 321 |
+
| | | heard as | errors on the line |
|
| 322 |
+
|---|---|---|---|
|
| 323 |
+
| **32B** | `m a ʀ s ɛ j` | *Marseille* | — |
|
| 324 |
+
| `8b-v3.1` | `m a ʀ s ɛ j` | *Marseille* | 4 / 71 |
|
| 325 |
+
| `4b-v3.1` | `m a ʀ s ɛ ʀ ɛ` | *"marcerre"* | 5 / 71 |
|
| 326 |
+
| `4b-v3` | `m a z ɛ ʀ` | *"mazer"* | 8 / 71 |
|
| 327 |
+
|
| 328 |
+
Note how little the toponym costs: `4b-v3.1` botches the name outright and pays **one** phoneme more than `8b-v3.1` over the whole sentence. That is exactly why the aggregate scores cannot settle the proper-noun question, and why it gets its own section below rather than a place in the ranking.
|
| 329 |
+
|
| 330 |
+
## Per language, because the average hides everything
|
| 331 |
+
|
| 332 |
+
Errors are counted against the 32B **of the same seed**, averaged over the three seeds. The first two columns are the yardstick: how long the reference is, and how much the 32B differs from *itself*.
|
| 333 |
+
|
| 334 |
+
| | length | **threshold** | `8b-v3.1` | `8b-v3.1-mlp` | `4b-v3.1` | `4b-v3.1-mlp` |
|
| 335 |
+
|---|---|---|---|---|---|---|
|
| 336 |
+
| en | 77 | **0.0** | 0.7 | 0.7 | 0.7 | 0.0 |
|
| 337 |
+
| es | 62 | **0.0** | 1.3 | 0.0 | 0.7 | 0.0 |
|
| 338 |
+
| de | 97 | 2.7 | 3.0 | 5.0 | 3.7 | 2.7 |
|
| 339 |
+
| it | 62 | 2.3 | 0.7 | 0.3 | 4.3 | 1.0 |
|
| 340 |
+
| zh | 85 | 3.3 | 3.3 | 1.0 | 2.7 | 2.0 |
|
| 341 |
+
| fr | 70 | 4.0 | 1.7 | 1.3 | 5.0 | 4.0 |
|
| 342 |
+
| ar | 94 | 6.3 | 5.7 | 4.7 | 7.7 | 12.3 |
|
| 343 |
+
| ja | 73 | 6.7 | 6.3 | 6.7 | 5.7 | 7.0 |
|
| 344 |
+
| ru | 77 | 10.7 | 14.0 | 16.3 | 14.0 | 10.0 |
|
| 345 |
+
| ko | 64 | 11.3 | 11.0 | 15.0 | 10.7 | 9.3 |
|
| 346 |
+
| pt | 62 | **17.0** | 22.7 | 23.0 | 22.3 | 28.3 |
|
| 347 |
+
|
| 348 |
+
Read it against the threshold column, never in absolute terms:
|
| 349 |
+
|
| 350 |
+
- **English and Spanish** — the 32B repeats itself phoneme for phoneme. There, a single phoneme of drift is real signal, and all four files stay within one.
|
| 351 |
+
- **French, Italian, Chinese, Arabic, Japanese** — the projections are *at or below* the 32B's own variance. In French both 8B files land at 1.7 and 1.3 against a threshold of 4.0: closer to the 32B than the 32B is to itself.
|
| 352 |
+
- **Portuguese, Russian and Korean** carry thresholds of 17.0, 10.7 and 11.3 — the reference rewrites a large share of its own pronunciation between seeds. Any single-seed comparison there was measuring the dice.
|
| 353 |
+
|
| 354 |
+
### Where the phoneme metric misleads, and the cross-check that catches it
|
| 355 |
+
|
| 356 |
+
A high threshold does not mean the speech is bad. It means **the phoneme transcriber cannot hold that
|
| 357 |
+
language still.** Cross-checking against Whisper, which reads words rather than sounds, on the same
|
| 358 |
+
renders:
|
| 359 |
+
|
| 360 |
+
| | ZIPA threshold | ZIPA v3.1 | × threshold | **Whisper, 32B** | **Whisper, v3.1** |
|
| 361 |
+
|---|---|---|---|---|---|
|
| 362 |
+
| pt | 17.0 | 24.1 | **1.42** | **0 / 88** | **1.8 / 88** |
|
| 363 |
+
| ru | 10.7 | 13.6 | 1.27 | **0 / 85** | 2.8 / 85 |
|
| 364 |
+
| ko | 11.3 | 11.5 | 1.01 | 2 / 85 | 3.2 / 85 |
|
| 365 |
+
| ja | 6.7 | 6.4 | 0.96 | 2 / 39 | 5.2 / 39 |
|
| 366 |
+
| zh | 3.3 | 2.2 | 0.67 | 2 / 31 | 3.5 / 31 |
|
| 367 |
+
|
| 368 |
+
**Portuguese is the worst language by phoneme and one of the best by word** — zero character errors for
|
| 369 |
+
the 32B, 2 % for the v3.1 files. Russian likewise: Whisper transcribes the 32B and two of the projections
|
| 370 |
+
word for word.
|
| 371 |
+
|
| 372 |
+
The cause is exactly what makes ZIPA useful elsewhere: it has no lexical decoder. European Portuguese
|
| 373 |
+
elides and reduces its vowels, Russian has vowel reduction under stress shift — the phonetic realisation
|
| 374 |
+
moves from one draw to the next while the word does not. ZIPA counts every allophonic variation as an
|
| 375 |
+
error; Whisper, which recognises the word, sees none. In Japanese and Chinese the bias runs the other
|
| 376 |
+
way: Whisper is harsher, because one missed ideogram weighs heavily on 31 characters.
|
| 377 |
+
|
| 378 |
+
**Neither metric is sufficient alone.** Where the ZIPA threshold is high, read the word column.
|
| 379 |
+
(Whisper figures are single-seed, on the earlier 0.3 MP protocol.)
|
| 380 |
+
|
| 381 |
+
---
|
| 382 |
+
|
| 383 |
+
## The 32B is one of the least stable models here
|
| 384 |
+
|
| 385 |
+
Distance between two renders of the **same** model, seed changed, nothing else:
|
| 386 |
+
|
| 387 |
+
| | against itself | against the 32B |
|
| 388 |
+
|---|---|---|
|
| 389 |
+
| `4b-v3.1` | **3.6** | 7.0 |
|
| 390 |
+
| `4b-v3.1-mlp` | 4.0 | 7.0 |
|
| 391 |
+
| `8b-v3.1` | 4.1 | 6.4 |
|
| 392 |
+
| `8b-v3.1-mlp` | 4.7 | 6.7 |
|
| 393 |
+
| **32B** | **5.8** | — |
|
| 394 |
+
|
| 395 |
+
The projections repeat themselves *better* than the model they imitate.
|
| 396 |
+
|
| 397 |
+
**And the gap to the 32B is reproducible, not random.** Each projection sits far closer to itself (3.6–4.7) than to the 32B (6.4–7.0). If swapping the encoder merely added randomness, those two columns would match. They do not — each file redoes the same offset on every seed.
|
| 398 |
+
|
| 399 |
+
**It is not an accent either.** An accent would mean one phoneme consistently rendered as another. Counting the actual substitutions says otherwise:
|
| 400 |
+
|
| 401 |
+
| | substitutions | covered by recurring patterns |
|
| 402 |
+
|---|---|---|
|
| 403 |
+
| **32B against itself** | **103** | ɑ→a ×10, ɾ→r ×6, ʒ→ʐ ×5 |
|
| 404 |
+
| `8b-v3.1` | **103** | 4 % — one pattern |
|
| 405 |
+
| `4b-v3.1` | 114 | **0 %** |
|
| 406 |
+
| `8b-v3.1-mlp` | 115 | 8 % |
|
| 407 |
+
| `4b-v3.1-mlp` | 125 | **0 %** |
|
| 408 |
+
| the four v3 files | 153–196 | 2–13 % |
|
| 409 |
+
|
| 410 |
+
The v3.1 files produce **as many substitutions as the 32B inflicts on itself** — 103 to 125 against 103 — and almost none of them form a repeating pattern. The offset is reproducible but scattered across many different sounds rather than concentrated into a signature. Ironically the clearest patterns belong to the 32B itself, between its own seeds, where they are ordinary allophonic variation.
|
| 411 |
+
|
| 412 |
+
The v3 files produce 1.5 to 2 times as many.
|
| 413 |
+
|
| 414 |
+
None of which tells you what it *sounds* like. A native speaker might well hear something none of these counts describe.
|
| 415 |
+
|
| 416 |
+
---
|
| 417 |
+
|
| 418 |
+
## An anecdote, and why it is not a result
|
| 419 |
+
|
| 420 |
+
**This measures nothing.** One word, in one language out of eleven, with no denominator — it is recorded
|
| 421 |
+
here because it was noticed, not because it supports a conclusion. It is deliberately absent from every
|
| 422 |
+
table above.
|
| 423 |
+
|
| 424 |
+
The French prompt contains a city name. The four 4B files render it as a non-word on almost every seed,
|
| 425 |
+
the 8B files render it correctly on all three:
|
| 426 |
+
|
| 427 |
+
| | seed 42 | seed 100 k | seed 100 M |
|
| 428 |
+
|---|---|---|---|
|
| 429 |
+
| 32B | ✓ | ✓ | ✓ |
|
| 430 |
+
| `8b-v3.1` | ✓ | ✓ | ✓ |
|
| 431 |
+
| `8b-v3.1-mlp` | ✓ | ✓ | ✓ |
|
| 432 |
+
| `4b-v3.1-mlp` | ✗ | ✗ | ✓ |
|
| 433 |
+
| `4b-v3.1` | ✗ | ✗ | ✗ |
|
| 434 |
+
|
| 435 |
+
What keeps it from being a finding, beyond the sample size: **it costs almost nothing on the sentence.**
|
| 436 |
+
`4b-v3.1` mangles the name outright and ends up **one** phoneme worse than `8b-v3.1` over the whole line.
|
| 437 |
+
Every aggregate metric in this report is blind to it, which cuts both ways — they cannot confirm it either.
|
| 438 |
+
|
| 439 |
+
The aggregate scores say 4B and 8B are equivalent, and that is the conclusion to keep. The only reason to
|
| 440 |
+
mention this at all is that quantisation is independently known to cost factual recall: if your prompts
|
| 441 |
+
lean on names of people or places, test both sizes on **your** prompts rather than trusting anything here.
|
| 442 |
+
|
| 443 |
+
---
|
| 444 |
+
|
| 445 |
+
## Which one to take
|
| 446 |
+
|
| 447 |
+
| If you | Take |
|
| 448 |
+
|---|---|
|
| 449 |
+
| generate images or video without speech | **`4b-ClipProj-v3.1`** — 4.6 GB, indistinguishable from the 32B |
|
| 450 |
+
| are tight on VRAM | **`4b-ClipProj-v3.1`** — the ridge is 26 MB and gives up nothing measurable |
|
| 451 |
+
| generate multilingual speech | **`4b-ClipProj-v3.1`** covers nine of the eleven languages tested; `8b-ClipProj-v3.1` has the best overall speech score, by less than the seed-to-seed spread |
|
| 452 |
+
| have the VRAM to spare | `8b-ClipProj-v3.1` — nothing measured says you need it, nothing says it hurts |
|
| 453 |
+
|
| 454 |
+
**Do not take a v3.** That is the only difference this benchmark resolves cleanly: v3 versus v3.1 is real,
|
| 455 |
+
4B versus 8B is not, ridge versus residual is not.
|
| 456 |
+
|
| 457 |
+
---
|
| 458 |
+
|
| 459 |
+
## How the speech benchmark got affordable
|
| 460 |
+
|
| 461 |
+
The old protocol rendered full 0.3 MP video and threw the picture away. Measured, same prompt and seed:
|
| 462 |
+
|
| 463 |
+
| | time |
|
| 464 |
+
|---|---|
|
| 465 |
+
| 0.3 MP video + audio + previews *(old)* | 80.3 s |
|
| 466 |
+
| same, video VAE and previews removed | 48.7 s |
|
| 467 |
+
| **64×64, 8 steps, audio only** | **12.4 s** |
|
| 468 |
+
|
| 469 |
+
Verified lossless before adopting: **+0.2 dB** across every octave band and **1 phoneme out of 70** for dropping the video decode; 64×64 costs 3 phonemes out of 70 against the full-resolution render.
|
| 470 |
+
|
| 471 |
+
**128×128 was rejected** — it truncates the start of the sentence, exactly the same eight phonemes at 6 steps and at 8. 64×64 does not. Counter-intuitive, reproducible, and the reason the whole benchmark runs at the smaller size.
|
| 472 |
+
|
| 473 |
+
That is what made three seeds across 297 renders possible at all: 31 minutes on two cards instead of six and a half hours.
|
| 474 |
+
|
| 475 |
+
---
|
| 476 |
+
|
| 477 |
+
## Limitations
|
| 478 |
+
|
| 479 |
+
**Three seeds fix the order of magnitude of the noise, not its tail.** Any gap under one point of score is not a result.
|
| 480 |
+
|
| 481 |
+
**The cosine is blind to countable attributes.** A whole loaf and a halved loaf, same crust, same paper, same light, give the same vector to the fourth decimal. Image equivalence here means *global appearance*, not attribute-by-attribute conformity.
|
| 482 |
+
|
| 483 |
+
**Speech quality is deliberately poor.** Six to eight steps gives a tinny, canned sound — identically for the 32B, with the same 19 dB dip between 1 and 3 kHz. The benchmark measures **correctness of pronunciation, not fidelity of reproduction**.
|
| 484 |
+
|
| 485 |
+
**The phoneme metric is unreliable in Portuguese, Russian and Korean** — not the speech itself. The
|
| 486 |
+
reference drifts by 17.0, 10.7 and 11.3 phonemes there between seeds, while Whisper transcribes the same
|
| 487 |
+
renders with zero to three character errors. Read the word column in those languages.
|
| 488 |
+
|
| 489 |
+
**Quantisation costs facts.** Known before, still true, and the most likely explanation for the proper-noun gap.
|
| 490 |
+
|
| 491 |
+
---
|
| 492 |
+
|
| 493 |
+
## Licence and responsibility
|
| 494 |
+
|
| 495 |
+
MIT, like the node. These matrices are derived from the activations of both models and their legal status is unclear; they are provided as-is, for research.
|
| 496 |
+
|
| 497 |
+
- **Qwen3-VL** — Alibaba, Apache 2.0.
|
| 498 |
+
- **MiniMax H3** — custom licence, read it before any commercial use.
|
| 499 |
+
|
| 500 |
+
Not affiliated with, endorsed by, or connected to Alibaba / Qwen, MiniMax, or Comfy Org. You remain responsible for what you generate.
|
| 501 |
+
|
| 502 |
+
---
|
| 503 |
+
|
| 504 |
+
## Credits
|
| 505 |
+
|
| 506 |
+
Vibe-coded with **Anthropic Claude Code (Opus 5)**. Every number here was measured on this hardware, never estimated. Where a prediction lost to a measurement, the measurement won and the text was rewritten — which happened three times in this release, the largest being a single-seed ranking of the v3.1 files that dissolved entirely once the threshold was known.
|
bench3.1/audio/ar/32b_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:57b11418c1f77eae19d2babf89fc1fded73b0e3ef59535892d0f5f0514764d8b
|
| 3 |
+
size 485541
|
bench3.1/audio/ar/32b_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ab9a979b8139deea2222bb98ed3490893e5202a2cc06c7f42a4c8bc02515c80c
|
| 3 |
+
size 452650
|
bench3.1/audio/ar/32b_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3a3f5f0ef60a068367644c5aa411fcc0a695635cfcc37071aa2f0fa6c3234b8e
|
| 3 |
+
size 488148
|
bench3.1/audio/ar/4b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5a27b47f79fdd7305286a3b8e07e3bb596758a8ab3e7109571af074a2d8c2e8c
|
| 3 |
+
size 467032
|
bench3.1/audio/ar/4b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ec82f202d8140f32d473d4090ae6810d48b6d4800c100a87673492eb80ee9962
|
| 3 |
+
size 459884
|
bench3.1/audio/ar/4b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:94e525bca63fee7add1ae785ed858be21199e940c5dcd86cf732e5e23a2b4e20
|
| 3 |
+
size 458763
|
bench3.1/audio/ar/4b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b13d863e50eac214ac6c1c57fb4a9ebdf4daffd433e7415a78b2070462e839ea
|
| 3 |
+
size 468000
|
bench3.1/audio/ar/4b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e11ccff921b504b3343fa24664b8006b41c205af367199515887c6c27e9b7958
|
| 3 |
+
size 470773
|
bench3.1/audio/ar/4b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:770f05d8a34d893f495874d9d8f3fefaa26af704d044edccc801a0c73d119cae
|
| 3 |
+
size 456558
|
bench3.1/audio/ar/4b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:34d03be6563937655a212e23f9d07258dc8a8a2aa34cf036ba262d3185ba90b3
|
| 3 |
+
size 477859
|
bench3.1/audio/ar/4b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:825851a2259a4181377aa0648c25d3f253c4135e55f47b9ce2c525adbc9e89f1
|
| 3 |
+
size 471423
|
bench3.1/audio/ar/4b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0cfd3f9a1e8f3b861503a428f8311ddff68fdbeb1785b007a97be9628fa4f421
|
| 3 |
+
size 486703
|
bench3.1/audio/ar/4b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:532d9d911f3753587c97c69f22127097c855317fbf4cc3c2610724e65c0a7937
|
| 3 |
+
size 484528
|
bench3.1/audio/ar/4b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e95fef01962c8155082b879d07bcd981800b0af20e720b210bb44cf3007c402e
|
| 3 |
+
size 466523
|
bench3.1/audio/ar/4b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3e323f0552b09566fdf7fe7dade15753e8afbc7a6a61699cf80a81b8af64323c
|
| 3 |
+
size 476437
|
bench3.1/audio/ar/8b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:46230a2d96de6707d170fe600178d98e44cd7bd46d5c199c0b56cab15d55f49b
|
| 3 |
+
size 470495
|
bench3.1/audio/ar/8b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2e416d4942d8d2f83a401d8432867bc243fbff2a3b3683968576670a36c9966f
|
| 3 |
+
size 459569
|
bench3.1/audio/ar/8b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9cbc8e090a142f33b66de725af3d9f925ef7c62769cd6701897676ea1955ddd2
|
| 3 |
+
size 465130
|
bench3.1/audio/ar/8b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9da4ba292d06e36a792ec328bbe89f6eb70769b94f4cdea79552d8dad169e3f3
|
| 3 |
+
size 478546
|
bench3.1/audio/ar/8b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2733da6c8b9603ea0323811ea81076b1f0dbe6e860d7627d4f9b17a4840f6d0c
|
| 3 |
+
size 457617
|
bench3.1/audio/ar/8b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d8d87110720ddc741a444b4a289b1b49206db72df7a4fd3f7c39de2f877414b6
|
| 3 |
+
size 476442
|
bench3.1/audio/ar/8b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:809f31b63afdaa6c663a2bd1983a7fcb30849842b245028918faeb8552213e1a
|
| 3 |
+
size 481752
|
bench3.1/audio/ar/8b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:56f60ee0468fa51640af552377d944f892f3b557de1a4ebdc08be218ff2e785b
|
| 3 |
+
size 469274
|
bench3.1/audio/ar/8b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ac31636b767be1cae7ac384f0f703ad7129b0eb3bc7bed1f5cfd078dce26a47e
|
| 3 |
+
size 491750
|
bench3.1/audio/ar/8b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:abdd4c7625b756495f85957509af9ce7b46236208c463f1822d7a89f9e6d5b67
|
| 3 |
+
size 482115
|
bench3.1/audio/ar/8b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fa3763570a075e993ac5d3f31c09c91ef883f5d7e8d23e5c4936646944e55742
|
| 3 |
+
size 463509
|
bench3.1/audio/ar/8b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7013b4ec3ec203164d0cd018c19b1f1fa58904c72bf9c278b0e2901d34b51b3c
|
| 3 |
+
size 490893
|
bench3.1/audio/de/32b_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:139088593e3c4a2c31477110c1484c28edff2df0da0e9462aad500a1cc250438
|
| 3 |
+
size 447679
|
bench3.1/audio/de/32b_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a7aa921e29ad449b432ec3970fdc1f66492f86597a641d1b067c1eaa1bbedf01
|
| 3 |
+
size 424947
|
bench3.1/audio/de/32b_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:afb1026d4b29af96fa59bebe332a4c2e455f16a4ed23495121525776627b8ce1
|
| 3 |
+
size 457866
|
bench3.1/audio/de/4b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e457d51449c8269939fcc7ad40f9b0e5ee9371523c6323f1cb055dfbae1064df
|
| 3 |
+
size 416507
|
bench3.1/audio/de/4b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4a1b60cd526837d6829c88b524382f86086c78b382b9add11d3bf2d788e9965f
|
| 3 |
+
size 383260
|
bench3.1/audio/de/4b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:40a145b8efd06cb83c34c11de97d32f516a88cf63c8fa6b70204051af7f56ed7
|
| 3 |
+
size 416276
|
bench3.1/audio/de/4b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9942736d876dc378d27af4fd30201cbf8abc2b62903d6a15e099c540dd4d87cb
|
| 3 |
+
size 424040
|
bench3.1/audio/de/4b-v3.1-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:59c34aebb926e9b7b85a93a698adbec0a24e7406ed4fceb30a32934e038bd1ae
|
| 3 |
+
size 409861
|
bench3.1/audio/de/4b-v3.1-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6d66d5a76ed2b232d768d87dfbcd80394f6762abac040e7ea322d47d87b80381
|
| 3 |
+
size 434627
|
bench3.1/audio/de/4b-v3.1_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:72637556a5a1dc77aa0bca0e5d3c449a6a6f0da784e6bddd1ea865a6a6a104d6
|
| 3 |
+
size 441075
|
bench3.1/audio/de/4b-v3.1_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:756e835c499eac85b49680623553ea4bd1da20d234c6678ab76e7d50a373df26
|
| 3 |
+
size 438930
|
bench3.1/audio/de/4b-v3.1_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:986c23484e2bd194cdee2b3a742c617f1c5ac172ec4930bf7491cc83ceb57c2e
|
| 3 |
+
size 457654
|
bench3.1/audio/de/4b-v3_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:97788cde2fdae7b3ba5044e8c78d46a080f12ceb1852a042110d062de39a17a9
|
| 3 |
+
size 440045
|
bench3.1/audio/de/4b-v3_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4f70eb8f148e6b419959ecba40610fcfcc9c2ee89eb5ff921fc1e2c464715081
|
| 3 |
+
size 427003
|
bench3.1/audio/de/4b-v3_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25be0a144b40341e7880ca96cc40d5cdd56088cc7a5126cd786c22520ed6838d
|
| 3 |
+
size 465126
|
bench3.1/audio/de/8b-v3-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a95292be2bff21e091c641965042d106bac04bb933f76a0679bd809863708a85
|
| 3 |
+
size 425098
|
bench3.1/audio/de/8b-v3-mlp_s100000000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3350935b192160ff5c724a6bded252c0f69a7d0c2d673f41abd1c21997c916b6
|
| 3 |
+
size 412601
|
bench3.1/audio/de/8b-v3-mlp_s42.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:475ff84db73b26e93e9c07ce01e92098fb2dab6873f80c7aaef7a96bafc69c6e
|
| 3 |
+
size 464571
|
bench3.1/audio/de/8b-v3.1-mlp_s100000.flac
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:89d21c09bfb7b348ac03ee8ee316e9b75a12867580b7b8e6919cde8c85017623
|
| 3 |
+
size 407809
|