sigargv commited on
Commit
3943f51
Β·
verified Β·
1 Parent(s): eb4fbd8

Add files using upload-large-folder tool

Browse files
.gitattributes CHANGED
@@ -53,3 +53,13 @@ IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00004-of-00010.gguf filter=lfs di
53
  IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00008-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
54
  IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00005-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
55
  IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00003-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
53
  IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00008-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
54
  IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00005-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
55
  IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00003-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
56
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00002-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
57
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00006-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
58
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
59
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00003-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
60
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00004-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
61
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00007-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
62
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00008-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
63
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00010-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
64
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00009-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
65
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00005-of-00010.gguf filter=lfs diff=lfs merge=lfs -text
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:95164a5592987dff50a9fad555b56c148b9d01ceff95816e2fa1175424151c60
3
+ size 11377052512
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00002-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b18ba2aafa656734013a325bbd5858fd3eecb026705094c904a96692cf7ca03
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00003-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d1cd6f3e2501029ce8c8ce82b6bbc21c3fd05f571070a7c3a6aca0969b3d4c9f
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00004-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4cee36fc0918c7fcb4ba728ff6790117e1beba642cec9b1104f6fab5ba2a6000
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00005-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2aa0b85f00f95c64107ce6a5e847d6c19291da3d931bb509f2ade52578b82fd1
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00006-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0390cc1e252052fb5e07e221c97a5cbb8c779483bd4706f0325b3818f51c1d94
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00007-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:63e90c41ea981b62ace5b1edc84096a3ba51d55db2759f68eb59786f4127a48c
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00008-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:27b8f023b41e0cc7f7a73955b84f0877b7ce6d01f288a16c631dac43c5aa63ba
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00009-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8311a8576b1a8d6ace41fa9adee21f14f01622dcfd959e08a27f839534b22edf
3
+ size 10150217440
IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00010-of-00010.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:649469aec90078da7b0e6263a3bed109769872c900e1f9176542d08e1d197647
3
+ size 5997855808
IQ3_S/SHA256SUMS ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ 95164a5592987dff50a9fad555b56c148b9d01ceff95816e2fa1175424151c60 Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf
2
+ 4b18ba2aafa656734013a325bbd5858fd3eecb026705094c904a96692cf7ca03 Laguna-M.1-IQ3_S-imatrix-public-v1-00002-of-00010.gguf
3
+ d1cd6f3e2501029ce8c8ce82b6bbc21c3fd05f571070a7c3a6aca0969b3d4c9f Laguna-M.1-IQ3_S-imatrix-public-v1-00003-of-00010.gguf
4
+ 4cee36fc0918c7fcb4ba728ff6790117e1beba642cec9b1104f6fab5ba2a6000 Laguna-M.1-IQ3_S-imatrix-public-v1-00004-of-00010.gguf
5
+ 2aa0b85f00f95c64107ce6a5e847d6c19291da3d931bb509f2ade52578b82fd1 Laguna-M.1-IQ3_S-imatrix-public-v1-00005-of-00010.gguf
6
+ 0390cc1e252052fb5e07e221c97a5cbb8c779483bd4706f0325b3818f51c1d94 Laguna-M.1-IQ3_S-imatrix-public-v1-00006-of-00010.gguf
7
+ 63e90c41ea981b62ace5b1edc84096a3ba51d55db2759f68eb59786f4127a48c Laguna-M.1-IQ3_S-imatrix-public-v1-00007-of-00010.gguf
8
+ 27b8f023b41e0cc7f7a73955b84f0877b7ce6d01f288a16c631dac43c5aa63ba Laguna-M.1-IQ3_S-imatrix-public-v1-00008-of-00010.gguf
9
+ 8311a8576b1a8d6ace41fa9adee21f14f01622dcfd959e08a27f839534b22edf Laguna-M.1-IQ3_S-imatrix-public-v1-00009-of-00010.gguf
10
+ 649469aec90078da7b0e6263a3bed109769872c900e1f9176542d08e1d197647 Laguna-M.1-IQ3_S-imatrix-public-v1-00010-of-00010.gguf
IQ3_S/eval-offload.tsv ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ phase ngl status log
2
+ smoke 99 ok /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/smoke.ngl99.log
3
+ ppl 99 ok /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/eval-ppl.ngl99.log
IQ3_S/eval-ppl.log ADDED
@@ -0,0 +1,276 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ main: build = 4641 (be7d53ce)
2
+ main: built with cc (Ubuntu 13.3.0-6ubuntu2~24.04.1) 13.3.0 for aarch64-linux-gnu
3
+ main: seed = 42
4
+ ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
5
+ ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
6
+ ggml_cuda_init: found 1 CUDA devices:
7
+ Device 0: NVIDIA GB10, compute capability 12.1, VMM: yes, VRAM: 124610 MiB
8
+ CUDA0: using device CUDA0 - 96489 MiB free
9
+ llama_model_loader: additional 9 GGUFs metadata loaded.
10
+ llama_model_loader: loaded meta data with 59 key-value pairs and 1178 tensors from /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf (version GGUF V3 (latest))
11
+ llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
12
+ llama_model_loader: - kv 0: general.architecture str = laguna
13
+ llama_model_loader: - kv 1: general.type str = model
14
+ llama_model_loader: - kv 2: general.name str = Laguna M.1
15
+ llama_model_loader: - kv 3: general.size_label str = 256x11B
16
+ llama_model_loader: - kv 4: general.license str = apache-2.0
17
+ llama_model_loader: - kv 5: general.tags arr[str,6] = ["laguna-m.1", "vllm", "sglang", "bf1...
18
+ llama_model_loader: - kv 6: laguna.context_length u32 = 262144
19
+ llama_model_loader: - kv 7: laguna.embedding_length u32 = 4096
20
+ llama_model_loader: - kv 8: laguna.block_count u32 = 70
21
+ llama_model_loader: - kv 9: laguna.feed_forward_length u32 = 16384
22
+ llama_model_loader: - kv 10: laguna.attention.head_count arr[i32,70] = [64, 64, 64, 64, 64, 64, 64, 64, 64, ...
23
+ llama_model_loader: - kv 11: laguna.attention.head_count_kv u32 = 8
24
+ llama_model_loader: - kv 12: laguna.attention.key_length u32 = 128
25
+ llama_model_loader: - kv 13: laguna.attention.value_length u32 = 128
26
+ llama_model_loader: - kv 14: laguna.attention.layer_norm_rms_epsilon f32 = 0.000001
27
+ llama_model_loader: - kv 15: general.file_type u32 = 26
28
+ llama_model_loader: - kv 16: laguna.attention.sliding_window u32 = 0
29
+ llama_model_loader: - kv 17: laguna.rope.dimension_count u32 = 128
30
+ llama_model_loader: - kv 18: laguna.rope.dimension_count_swa u32 = 128
31
+ llama_model_loader: - kv 19: laguna.rope.freq_base f32 = 500000.000000
32
+ llama_model_loader: - kv 20: laguna.rope.freq_base_swa f32 = 10000.000000
33
+ llama_model_loader: - kv 21: laguna.rope.scaling.type str = yarn
34
+ llama_model_loader: - kv 22: laguna.rope.scaling.factor f32 = 64.000000
35
+ llama_model_loader: - kv 23: laguna.rope.scaling.original_context_length u32 = 4096
36
+ llama_model_loader: - kv 24: laguna.rope.scaling.yarn_ext_factor f32 = 1.000000
37
+ llama_model_loader: - kv 25: laguna.rope.scaling.yarn_attn_factor f32 = 1.000000
38
+ llama_model_loader: - kv 26: laguna.rope.scaling.yarn_beta_fast f32 = 64.000000
39
+ llama_model_loader: - kv 27: laguna.rope.scaling.yarn_beta_slow f32 = 1.000000
40
+ llama_model_loader: - kv 28: laguna.expert_count u32 = 256
41
+ llama_model_loader: - kv 29: laguna.expert_used_count u32 = 16
42
+ llama_model_loader: - kv 30: laguna.expert_feed_forward_length u32 = 1024
43
+ llama_model_loader: - kv 31: laguna.expert_shared_feed_forward_length u32 = 1024
44
+ llama_model_loader: - kv 32: laguna.expert_weights_scale f32 = 1.000000
45
+ llama_model_loader: - kv 33: laguna.expert_weights_norm bool = true
46
+ llama_model_loader: - kv 34: laguna.expert_gating_func u32 = 2
47
+ llama_model_loader: - kv 35: laguna.leading_dense_block_count u32 = 3
48
+ llama_model_loader: - kv 36: tokenizer.ggml.model str = gpt2
49
+ llama_model_loader: - kv 37: tokenizer.ggml.pre str = laguna
50
+ llama_model_loader: - kv 38: tokenizer.ggml.tokens arr[str,100352] = ["γ€ˆ|UNK|〉", "γ€ˆ|CODE_START|〉",...
51
+ llama_model_loader: - kv 39: tokenizer.ggml.token_type arr[i32,100352] = [3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, ...
52
+ llama_model_loader: - kv 40: tokenizer.ggml.merges arr[str,100026] = ["i n", "Δ  t", "Δ  Δ ", "e r", "Δ  a...
53
+ llama_model_loader: - kv 41: tokenizer.ggml.bos_token_id u32 = 2
54
+ llama_model_loader: - kv 42: tokenizer.ggml.eos_token_id u32 = 2
55
+ llama_model_loader: - kv 43: tokenizer.ggml.unknown_token_id u32 = 0
56
+ llama_model_loader: - kv 44: tokenizer.ggml.seperator_token_id u32 = 8
57
+ llama_model_loader: - kv 45: tokenizer.ggml.padding_token_id u32 = 9
58
+ llama_model_loader: - kv 46: tokenizer.ggml.mask_token_id u32 = 12
59
+ llama_model_loader: - kv 47: tokenizer.ggml.add_bos_token bool = true
60
+ llama_model_loader: - kv 48: tokenizer.ggml.add_sep_token bool = true
61
+ llama_model_loader: - kv 49: tokenizer.chat_template str = {#- Copied from laguna_glm_thinking_v...
62
+ llama_model_loader: - kv 50: tokenizer.ggml.eot_token_id u32 = 24
63
+ llama_model_loader: - kv 51: general.quantization_version u32 = 2
64
+ llama_model_loader: - kv 52: quantize.imatrix.file str = /mnt/pool/gguf/laguna-m1/imatrix/q8-p...
65
+ llama_model_loader: - kv 53: quantize.imatrix.dataset str = /workspace/laguna-m1/corpus/laguna-m1...
66
+ llama_model_loader: - kv 54: quantize.imatrix.entries_count i32 = 828
67
+ llama_model_loader: - kv 55: quantize.imatrix.chunks_count i32 = 2048
68
+ llama_model_loader: - kv 56: split.no u16 = 0
69
+ llama_model_loader: - kv 57: split.count u16 = 10
70
+ llama_model_loader: - kv 58: split.tensors.count i32 = 1178
71
+ llama_model_loader: - type f32: 415 tensors
72
+ llama_model_loader: - type q8_0: 140 tensors
73
+ llama_model_loader: - type q5_K: 70 tensors
74
+ llama_model_loader: - type q6_K: 2 tensors
75
+ llama_model_loader: - type iq3_s: 551 tensors
76
+ load: 0 unused tokens
77
+ load: special_eos_id is not in special_eog_ids - the tokenizer config may be incorrect
78
+ load: special_eot_id is not in special_eog_ids - the tokenizer config may be incorrect
79
+ load: printing all EOG tokens:
80
+ load: - 2 ('γ€ˆ|EOS|〉')
81
+ load: - 24 ('</assistant>')
82
+ load: special tokens cache size = 70
83
+ load: token to piece cache size = 0.6432 MB
84
+ llm_load_print_meta: format = GGUF V3 (latest)
85
+ llm_load_print_meta: arch = laguna
86
+ llm_load_print_meta: n_ctx_train = 262144
87
+ llm_load_print_meta: n_embd = 4096
88
+ llm_load_print_meta: n_layer = 70
89
+ llm_load_print_meta: n_head = 64
90
+ llm_load_print_meta: n_head_kv = 8
91
+ llm_load_print_meta: n_rot = 128
92
+ llm_load_print_meta: n_swa = 0
93
+ llm_load_print_meta: n_swa_pattern = 1
94
+ llm_load_print_meta: n_embd_head_k = 128
95
+ llm_load_print_meta: n_embd_head_v = 128
96
+ llm_load_print_meta: n_gqa = 8
97
+ llm_load_print_meta: n_embd_k_gqa = 1024
98
+ llm_load_print_meta: n_embd_v_gqa = 1024
99
+ llm_load_print_meta: f_norm_eps = 0.0e+00
100
+ llm_load_print_meta: f_norm_rms_eps = 1.0e-06
101
+ llm_load_print_meta: f_clamp_kqv = 0.0e+00
102
+ llm_load_print_meta: f_max_alibi_bias = 0.0e+00
103
+ llm_load_print_meta: f_logit_scale = 0.0e+00
104
+ llm_load_print_meta: n_ff = 16384
105
+ llm_load_print_meta: n_expert = 256
106
+ llm_load_print_meta: n_expert_used = 16
107
+ llm_load_print_meta: causal attn = 1
108
+ llm_load_print_meta: pooling type = 0
109
+ llm_load_print_meta: rope type = 2
110
+ llm_load_print_meta: rope scaling = yarn
111
+ llm_load_print_meta: freq_base_train = 500000.0
112
+ llm_load_print_meta: freq_scale_train = 0.015625
113
+ llm_load_print_meta: n_ctx_orig_yarn = 4096
114
+ llm_load_print_meta: rope_finetuned = unknown
115
+ llm_load_print_meta: ssm_d_conv = 0
116
+ llm_load_print_meta: ssm_d_inner = 0
117
+ llm_load_print_meta: ssm_d_state = 0
118
+ llm_load_print_meta: ssm_dt_rank = 0
119
+ llm_load_print_meta: ssm_n_group = 0
120
+ llm_load_print_meta: model type = ?B
121
+ llm_load_print_meta: model ftype = IQ3_S - 3.4375 bpw
122
+ llm_load_print_meta: model params = 225.796 B
123
+ llm_load_print_meta: model size = 91.803 GiB (3.492 BPW)
124
+ llm_load_print_meta: repeating layers = 91.175 GiB (3.481 BPW, 224.974 B parameters)
125
+ llm_load_print_meta: general.name = Laguna M.1
126
+ print_info: vocab type = BPE
127
+ print_info: n_vocab = 100352
128
+ print_info: n_merges = 100026
129
+ print_info: BOS token = 2 'γ€ˆ|EOS|〉'
130
+ print_info: EOS token = 2 'γ€ˆ|EOS|〉'
131
+ print_info: EOT token = 24 '</assistant>'
132
+ print_info: UNK token = 0 'γ€ˆ|UNK|〉'
133
+ print_info: SEP token = 8 'γ€ˆ|SEP|〉'
134
+ print_info: PAD token = 9 'γ€ˆ|PAD|〉'
135
+ print_info: MASK token = 12 'γ€ˆ|MASK|〉'
136
+ print_info: LF token = 268 'Ċ'
137
+ print_info: EOG token = 2 'γ€ˆ|EOS|〉'
138
+ print_info: EOG token = 24 '</assistant>'
139
+ print_info: max token length = 830
140
+ ======================================= HAVE_FANCY_SIMD is NOT defined
141
+ ------------------- Layer sizes:
142
+ Layer 0: 140.53, 16.00, 156.53 22.00 MiB
143
+ Layer 1: 140.53, 16.00, 156.53 22.00 MiB
144
+ Layer 2: 140.53, 16.00, 156.53 22.00 MiB
145
+ Layer 3: 1387.19, 16.00, 1403.19 52.00 MiB
146
+ Layer 4: 1387.19, 16.00, 1403.19 52.00 MiB
147
+ Layer 5: 1387.19, 16.00, 1403.19 52.00 MiB
148
+ Layer 6: 1387.19, 16.00, 1403.19 52.00 MiB
149
+ Layer 7: 1387.19, 16.00, 1403.19 52.00 MiB
150
+ Layer 8: 1387.19, 16.00, 1403.19 52.00 MiB
151
+ Layer 9: 1387.19, 16.00, 1403.19 52.00 MiB
152
+ Layer 10: 1387.19, 16.00, 1403.19 52.00 MiB
153
+ Layer 11: 1387.19, 16.00, 1403.19 52.00 MiB
154
+ Layer 12: 1387.19, 16.00, 1403.19 52.00 MiB
155
+ Layer 13: 1387.19, 16.00, 1403.19 52.00 MiB
156
+ Layer 14: 1387.19, 16.00, 1403.19 52.00 MiB
157
+ Layer 15: 1387.19, 16.00, 1403.19 52.00 MiB
158
+ Layer 16: 1387.19, 16.00, 1403.19 52.00 MiB
159
+ Layer 17: 1387.19, 16.00, 1403.19 52.00 MiB
160
+ Layer 18: 1387.19, 16.00, 1403.19 52.00 MiB
161
+ Layer 19: 1387.19, 16.00, 1403.19 52.00 MiB
162
+ Layer 20: 1387.19, 16.00, 1403.19 52.00 MiB
163
+ Layer 21: 1387.19, 16.00, 1403.19 52.00 MiB
164
+ Layer 22: 1387.19, 16.00, 1403.19 52.00 MiB
165
+ Layer 23: 1387.19, 16.00, 1403.19 52.00 MiB
166
+ Layer 24: 1387.19, 16.00, 1403.19 52.00 MiB
167
+ Layer 25: 1387.19, 16.00, 1403.19 52.00 MiB
168
+ Layer 26: 1387.19, 16.00, 1403.19 52.00 MiB
169
+ Layer 27: 1387.19, 16.00, 1403.19 52.00 MiB
170
+ Layer 28: 1387.19, 16.00, 1403.19 52.00 MiB
171
+ Layer 29: 1387.19, 16.00, 1403.19 52.00 MiB
172
+ Layer 30: 1387.19, 16.00, 1403.19 52.00 MiB
173
+ Layer 31: 1387.19, 16.00, 1403.19 52.00 MiB
174
+ Layer 32: 1387.19, 16.00, 1403.19 52.00 MiB
175
+ Layer 33: 1387.19, 16.00, 1403.19 52.00 MiB
176
+ Layer 34: 1387.19, 16.00, 1403.19 52.00 MiB
177
+ Layer 35: 1387.19, 16.00, 1403.19 52.00 MiB
178
+ Layer 36: 1387.19, 16.00, 1403.19 52.00 MiB
179
+ Layer 37: 1387.19, 16.00, 1403.19 52.00 MiB
180
+ Layer 38: 1387.19, 16.00, 1403.19 52.00 MiB
181
+ Layer 39: 1387.19, 16.00, 1403.19 52.00 MiB
182
+ Layer 40: 1387.19, 16.00, 1403.19 52.00 MiB
183
+ Layer 41: 1387.19, 16.00, 1403.19 52.00 MiB
184
+ Layer 42: 1387.19, 16.00, 1403.19 52.00 MiB
185
+ Layer 43: 1387.19, 16.00, 1403.19 52.00 MiB
186
+ Layer 44: 1387.19, 16.00, 1403.19 52.00 MiB
187
+ Layer 45: 1387.19, 16.00, 1403.19 52.00 MiB
188
+ Layer 46: 1387.19, 16.00, 1403.19 52.00 MiB
189
+ Layer 47: 1387.19, 16.00, 1403.19 52.00 MiB
190
+ Layer 48: 1387.19, 16.00, 1403.19 52.00 MiB
191
+ Layer 49: 1387.19, 16.00, 1403.19 52.00 MiB
192
+ Layer 50: 1387.19, 16.00, 1403.19 52.00 MiB
193
+ Layer 51: 1387.19, 16.00, 1403.19 52.00 MiB
194
+ Layer 52: 1387.19, 16.00, 1403.19 52.00 MiB
195
+ Layer 53: 1387.19, 16.00, 1403.19 52.00 MiB
196
+ Layer 54: 1387.19, 16.00, 1403.19 52.00 MiB
197
+ Layer 55: 1387.19, 16.00, 1403.19 52.00 MiB
198
+ Layer 56: 1387.19, 16.00, 1403.19 52.00 MiB
199
+ Layer 57: 1387.19, 16.00, 1403.19 52.00 MiB
200
+ Layer 58: 1387.19, 16.00, 1403.19 52.00 MiB
201
+ Layer 59: 1387.19, 16.00, 1403.19 52.00 MiB
202
+ Layer 60: 1387.19, 16.00, 1403.19 52.00 MiB
203
+ Layer 61: 1387.19, 16.00, 1403.19 52.00 MiB
204
+ Layer 62: 1387.19, 16.00, 1403.19 52.00 MiB
205
+ Layer 63: 1387.19, 16.00, 1403.19 52.00 MiB
206
+ Layer 64: 1387.19, 16.00, 1403.19 52.00 MiB
207
+ Layer 65: 1387.19, 16.00, 1403.19 52.00 MiB
208
+ Layer 66: 1387.19, 16.00, 1403.19 52.00 MiB
209
+ Layer 67: 1387.19, 16.00, 1403.19 52.00 MiB
210
+ Layer 68: 1387.19, 16.00, 1403.19 52.00 MiB
211
+ Layer 69: 1387.19, 16.00, 1403.19 52.00 MiB
212
+ Layer 70: 321.56, 0.00, 321.56 MiB (output layer)
213
+ --------------------------------------------------------------------------
214
+ Total : 93363.29, 1120.00, 94483.29 MiB
215
+ Memory required for model tensors + cache: 94805 MiB
216
+ Memory available on all devices - compute: 95379 MiB
217
+ llm_load_tensors: ggml ctx size = 1.02 MiB
218
+ llm_load_tensors: offloading 70 repeating layers to GPU
219
+ llm_load_tensors: offloading non-repeating layers to GPU
220
+ llm_load_tensors: offloaded 71/71 layers to GPU
221
+ llm_load_tensors: CPU buffer size = 321.56 MiB
222
+ llm_load_tensors: CUDA0 buffer size = 93684.87 MiB
223
+ ....................................................................................................
224
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
225
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
226
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
227
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
228
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
229
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
230
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
231
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
232
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
233
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
234
+ llm_load_tensors: dense parameters loaded in 110.62s (5.44 GiB), expert parameters deferred (86.37 GiB)
235
+ llama_init_from_model: n_ctx = 4096
236
+ llama_init_from_model: n_batch = 512
237
+ llama_init_from_model: n_ubatch = 128
238
+ llama_init_from_model: flash_attn = 1
239
+ llama_init_from_model: attn_max_b = 0
240
+ llama_init_from_model: fused_moe = 1
241
+ llama_init_from_model: grouped er = 0
242
+ llama_init_from_model: fused_up_gate = 1
243
+ llama_init_from_model: fused_mmad = 1
244
+ llama_init_from_model: rope_cache = 0
245
+ llama_init_from_model: graph_reuse = 1
246
+ llama_init_from_model: k_cache_hadam = 0
247
+ llama_init_from_model: v_cache_hadam = 0
248
+ llama_init_from_model: split_mode_graph_scheduling = 0
249
+ llama_init_from_model: reduce_type = f16
250
+ llama_init_from_model: sched_async = 0
251
+ llama_init_from_model: ser = -1, 0
252
+ llama_init_from_model: freq_base = 500000.0
253
+ llama_init_from_model: freq_scale = 0.015625
254
+ llama_kv_cache_init: CUDA0 KV buffer size = 1120.00 MiB
255
+ llama_init_from_model: KV self size = 1120.00 MiB, K (f16): 560.00 MiB, V (f16): 560.00 MiB
256
+ llama_init_from_model: CUDA_Host output buffer size = 0.38 MiB
257
+ llama_init_from_model: CUDA0 compute buffer size = 51.00 MiB
258
+ llama_init_from_model: CUDA_Host compute buffer size = 3.00 MiB
259
+ llama_init_from_model: graph nodes = 3174
260
+ llama_init_from_model: graph splits = 2
261
+ llama_init_from_model: enabling only_active_experts scheduling
262
+
263
+ system_info: n_threads = 20 / 20 | AVX = 0 | AVX_VNNI = 0 | AVX2 = 0 | AVX512 = 0 | AVX512_VBMI = 0 | AVX512_VNNI = 0 | AVX512_BF16 = 0 | FMA = 0 | NEON = 1 | SVE = 0 | ARM_FMA = 1 | F16C = 0 | FP16_VA = 1 | WASM_SIMD = 0 | SSE3 = 0 | SSSE3 = 0 | VSX = 0 | MATMUL_INT8 = 0 |
264
+ perplexity: tokenizing the input ..
265
+ perplexity: tokenization took 1789.63 ms
266
+ perplexity: calculating perplexity over 8 chunks, n_ctx=4096, batch_size=512, n_seq=1
267
+ perplexity: 25.98 seconds per pass - ETA 3.45 minutes
268
+ [1]1.3713,[2]2.2535,[3]1.8046,[4]2.4539,[5]2.6550,[6]2.6851,[7]2.6250,[8]2.4683,
269
+ llama_print_timings: load time = 128307.95 ms
270
+ llama_print_timings: sample time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
271
+ llama_print_timings: prompt eval time = 209390.13 ms / 32768 tokens ( 6.39 ms per token, 156.49 tokens per second)
272
+ llama_print_timings: eval time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
273
+ llama_print_timings: total time = 322931.50 ms / 32769 tokens
274
+ ~ggml_backend_cuda_context: have 1 graphs
275
+
276
+ Final estimate: PPL over 8 chunks for n_ctx=4096 = 2.4683 +/- 0.04051
IQ3_S/eval-ppl.ngl99.log ADDED
@@ -0,0 +1,276 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ main: build = 4641 (be7d53ce)
2
+ main: built with cc (Ubuntu 13.3.0-6ubuntu2~24.04.1) 13.3.0 for aarch64-linux-gnu
3
+ main: seed = 42
4
+ ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
5
+ ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
6
+ ggml_cuda_init: found 1 CUDA devices:
7
+ Device 0: NVIDIA GB10, compute capability 12.1, VMM: yes, VRAM: 124610 MiB
8
+ CUDA0: using device CUDA0 - 96489 MiB free
9
+ llama_model_loader: additional 9 GGUFs metadata loaded.
10
+ llama_model_loader: loaded meta data with 59 key-value pairs and 1178 tensors from /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf (version GGUF V3 (latest))
11
+ llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
12
+ llama_model_loader: - kv 0: general.architecture str = laguna
13
+ llama_model_loader: - kv 1: general.type str = model
14
+ llama_model_loader: - kv 2: general.name str = Laguna M.1
15
+ llama_model_loader: - kv 3: general.size_label str = 256x11B
16
+ llama_model_loader: - kv 4: general.license str = apache-2.0
17
+ llama_model_loader: - kv 5: general.tags arr[str,6] = ["laguna-m.1", "vllm", "sglang", "bf1...
18
+ llama_model_loader: - kv 6: laguna.context_length u32 = 262144
19
+ llama_model_loader: - kv 7: laguna.embedding_length u32 = 4096
20
+ llama_model_loader: - kv 8: laguna.block_count u32 = 70
21
+ llama_model_loader: - kv 9: laguna.feed_forward_length u32 = 16384
22
+ llama_model_loader: - kv 10: laguna.attention.head_count arr[i32,70] = [64, 64, 64, 64, 64, 64, 64, 64, 64, ...
23
+ llama_model_loader: - kv 11: laguna.attention.head_count_kv u32 = 8
24
+ llama_model_loader: - kv 12: laguna.attention.key_length u32 = 128
25
+ llama_model_loader: - kv 13: laguna.attention.value_length u32 = 128
26
+ llama_model_loader: - kv 14: laguna.attention.layer_norm_rms_epsilon f32 = 0.000001
27
+ llama_model_loader: - kv 15: general.file_type u32 = 26
28
+ llama_model_loader: - kv 16: laguna.attention.sliding_window u32 = 0
29
+ llama_model_loader: - kv 17: laguna.rope.dimension_count u32 = 128
30
+ llama_model_loader: - kv 18: laguna.rope.dimension_count_swa u32 = 128
31
+ llama_model_loader: - kv 19: laguna.rope.freq_base f32 = 500000.000000
32
+ llama_model_loader: - kv 20: laguna.rope.freq_base_swa f32 = 10000.000000
33
+ llama_model_loader: - kv 21: laguna.rope.scaling.type str = yarn
34
+ llama_model_loader: - kv 22: laguna.rope.scaling.factor f32 = 64.000000
35
+ llama_model_loader: - kv 23: laguna.rope.scaling.original_context_length u32 = 4096
36
+ llama_model_loader: - kv 24: laguna.rope.scaling.yarn_ext_factor f32 = 1.000000
37
+ llama_model_loader: - kv 25: laguna.rope.scaling.yarn_attn_factor f32 = 1.000000
38
+ llama_model_loader: - kv 26: laguna.rope.scaling.yarn_beta_fast f32 = 64.000000
39
+ llama_model_loader: - kv 27: laguna.rope.scaling.yarn_beta_slow f32 = 1.000000
40
+ llama_model_loader: - kv 28: laguna.expert_count u32 = 256
41
+ llama_model_loader: - kv 29: laguna.expert_used_count u32 = 16
42
+ llama_model_loader: - kv 30: laguna.expert_feed_forward_length u32 = 1024
43
+ llama_model_loader: - kv 31: laguna.expert_shared_feed_forward_length u32 = 1024
44
+ llama_model_loader: - kv 32: laguna.expert_weights_scale f32 = 1.000000
45
+ llama_model_loader: - kv 33: laguna.expert_weights_norm bool = true
46
+ llama_model_loader: - kv 34: laguna.expert_gating_func u32 = 2
47
+ llama_model_loader: - kv 35: laguna.leading_dense_block_count u32 = 3
48
+ llama_model_loader: - kv 36: tokenizer.ggml.model str = gpt2
49
+ llama_model_loader: - kv 37: tokenizer.ggml.pre str = laguna
50
+ llama_model_loader: - kv 38: tokenizer.ggml.tokens arr[str,100352] = ["γ€ˆ|UNK|〉", "γ€ˆ|CODE_START|〉",...
51
+ llama_model_loader: - kv 39: tokenizer.ggml.token_type arr[i32,100352] = [3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, ...
52
+ llama_model_loader: - kv 40: tokenizer.ggml.merges arr[str,100026] = ["i n", "Δ  t", "Δ  Δ ", "e r", "Δ  a...
53
+ llama_model_loader: - kv 41: tokenizer.ggml.bos_token_id u32 = 2
54
+ llama_model_loader: - kv 42: tokenizer.ggml.eos_token_id u32 = 2
55
+ llama_model_loader: - kv 43: tokenizer.ggml.unknown_token_id u32 = 0
56
+ llama_model_loader: - kv 44: tokenizer.ggml.seperator_token_id u32 = 8
57
+ llama_model_loader: - kv 45: tokenizer.ggml.padding_token_id u32 = 9
58
+ llama_model_loader: - kv 46: tokenizer.ggml.mask_token_id u32 = 12
59
+ llama_model_loader: - kv 47: tokenizer.ggml.add_bos_token bool = true
60
+ llama_model_loader: - kv 48: tokenizer.ggml.add_sep_token bool = true
61
+ llama_model_loader: - kv 49: tokenizer.chat_template str = {#- Copied from laguna_glm_thinking_v...
62
+ llama_model_loader: - kv 50: tokenizer.ggml.eot_token_id u32 = 24
63
+ llama_model_loader: - kv 51: general.quantization_version u32 = 2
64
+ llama_model_loader: - kv 52: quantize.imatrix.file str = /mnt/pool/gguf/laguna-m1/imatrix/q8-p...
65
+ llama_model_loader: - kv 53: quantize.imatrix.dataset str = /workspace/laguna-m1/corpus/laguna-m1...
66
+ llama_model_loader: - kv 54: quantize.imatrix.entries_count i32 = 828
67
+ llama_model_loader: - kv 55: quantize.imatrix.chunks_count i32 = 2048
68
+ llama_model_loader: - kv 56: split.no u16 = 0
69
+ llama_model_loader: - kv 57: split.count u16 = 10
70
+ llama_model_loader: - kv 58: split.tensors.count i32 = 1178
71
+ llama_model_loader: - type f32: 415 tensors
72
+ llama_model_loader: - type q8_0: 140 tensors
73
+ llama_model_loader: - type q5_K: 70 tensors
74
+ llama_model_loader: - type q6_K: 2 tensors
75
+ llama_model_loader: - type iq3_s: 551 tensors
76
+ load: 0 unused tokens
77
+ load: special_eos_id is not in special_eog_ids - the tokenizer config may be incorrect
78
+ load: special_eot_id is not in special_eog_ids - the tokenizer config may be incorrect
79
+ load: printing all EOG tokens:
80
+ load: - 2 ('γ€ˆ|EOS|〉')
81
+ load: - 24 ('</assistant>')
82
+ load: special tokens cache size = 70
83
+ load: token to piece cache size = 0.6432 MB
84
+ llm_load_print_meta: format = GGUF V3 (latest)
85
+ llm_load_print_meta: arch = laguna
86
+ llm_load_print_meta: n_ctx_train = 262144
87
+ llm_load_print_meta: n_embd = 4096
88
+ llm_load_print_meta: n_layer = 70
89
+ llm_load_print_meta: n_head = 64
90
+ llm_load_print_meta: n_head_kv = 8
91
+ llm_load_print_meta: n_rot = 128
92
+ llm_load_print_meta: n_swa = 0
93
+ llm_load_print_meta: n_swa_pattern = 1
94
+ llm_load_print_meta: n_embd_head_k = 128
95
+ llm_load_print_meta: n_embd_head_v = 128
96
+ llm_load_print_meta: n_gqa = 8
97
+ llm_load_print_meta: n_embd_k_gqa = 1024
98
+ llm_load_print_meta: n_embd_v_gqa = 1024
99
+ llm_load_print_meta: f_norm_eps = 0.0e+00
100
+ llm_load_print_meta: f_norm_rms_eps = 1.0e-06
101
+ llm_load_print_meta: f_clamp_kqv = 0.0e+00
102
+ llm_load_print_meta: f_max_alibi_bias = 0.0e+00
103
+ llm_load_print_meta: f_logit_scale = 0.0e+00
104
+ llm_load_print_meta: n_ff = 16384
105
+ llm_load_print_meta: n_expert = 256
106
+ llm_load_print_meta: n_expert_used = 16
107
+ llm_load_print_meta: causal attn = 1
108
+ llm_load_print_meta: pooling type = 0
109
+ llm_load_print_meta: rope type = 2
110
+ llm_load_print_meta: rope scaling = yarn
111
+ llm_load_print_meta: freq_base_train = 500000.0
112
+ llm_load_print_meta: freq_scale_train = 0.015625
113
+ llm_load_print_meta: n_ctx_orig_yarn = 4096
114
+ llm_load_print_meta: rope_finetuned = unknown
115
+ llm_load_print_meta: ssm_d_conv = 0
116
+ llm_load_print_meta: ssm_d_inner = 0
117
+ llm_load_print_meta: ssm_d_state = 0
118
+ llm_load_print_meta: ssm_dt_rank = 0
119
+ llm_load_print_meta: ssm_n_group = 0
120
+ llm_load_print_meta: model type = ?B
121
+ llm_load_print_meta: model ftype = IQ3_S - 3.4375 bpw
122
+ llm_load_print_meta: model params = 225.796 B
123
+ llm_load_print_meta: model size = 91.803 GiB (3.492 BPW)
124
+ llm_load_print_meta: repeating layers = 91.175 GiB (3.481 BPW, 224.974 B parameters)
125
+ llm_load_print_meta: general.name = Laguna M.1
126
+ print_info: vocab type = BPE
127
+ print_info: n_vocab = 100352
128
+ print_info: n_merges = 100026
129
+ print_info: BOS token = 2 'γ€ˆ|EOS|〉'
130
+ print_info: EOS token = 2 'γ€ˆ|EOS|〉'
131
+ print_info: EOT token = 24 '</assistant>'
132
+ print_info: UNK token = 0 'γ€ˆ|UNK|〉'
133
+ print_info: SEP token = 8 'γ€ˆ|SEP|〉'
134
+ print_info: PAD token = 9 'γ€ˆ|PAD|〉'
135
+ print_info: MASK token = 12 'γ€ˆ|MASK|〉'
136
+ print_info: LF token = 268 'Ċ'
137
+ print_info: EOG token = 2 'γ€ˆ|EOS|〉'
138
+ print_info: EOG token = 24 '</assistant>'
139
+ print_info: max token length = 830
140
+ ======================================= HAVE_FANCY_SIMD is NOT defined
141
+ ------------------- Layer sizes:
142
+ Layer 0: 140.53, 16.00, 156.53 22.00 MiB
143
+ Layer 1: 140.53, 16.00, 156.53 22.00 MiB
144
+ Layer 2: 140.53, 16.00, 156.53 22.00 MiB
145
+ Layer 3: 1387.19, 16.00, 1403.19 52.00 MiB
146
+ Layer 4: 1387.19, 16.00, 1403.19 52.00 MiB
147
+ Layer 5: 1387.19, 16.00, 1403.19 52.00 MiB
148
+ Layer 6: 1387.19, 16.00, 1403.19 52.00 MiB
149
+ Layer 7: 1387.19, 16.00, 1403.19 52.00 MiB
150
+ Layer 8: 1387.19, 16.00, 1403.19 52.00 MiB
151
+ Layer 9: 1387.19, 16.00, 1403.19 52.00 MiB
152
+ Layer 10: 1387.19, 16.00, 1403.19 52.00 MiB
153
+ Layer 11: 1387.19, 16.00, 1403.19 52.00 MiB
154
+ Layer 12: 1387.19, 16.00, 1403.19 52.00 MiB
155
+ Layer 13: 1387.19, 16.00, 1403.19 52.00 MiB
156
+ Layer 14: 1387.19, 16.00, 1403.19 52.00 MiB
157
+ Layer 15: 1387.19, 16.00, 1403.19 52.00 MiB
158
+ Layer 16: 1387.19, 16.00, 1403.19 52.00 MiB
159
+ Layer 17: 1387.19, 16.00, 1403.19 52.00 MiB
160
+ Layer 18: 1387.19, 16.00, 1403.19 52.00 MiB
161
+ Layer 19: 1387.19, 16.00, 1403.19 52.00 MiB
162
+ Layer 20: 1387.19, 16.00, 1403.19 52.00 MiB
163
+ Layer 21: 1387.19, 16.00, 1403.19 52.00 MiB
164
+ Layer 22: 1387.19, 16.00, 1403.19 52.00 MiB
165
+ Layer 23: 1387.19, 16.00, 1403.19 52.00 MiB
166
+ Layer 24: 1387.19, 16.00, 1403.19 52.00 MiB
167
+ Layer 25: 1387.19, 16.00, 1403.19 52.00 MiB
168
+ Layer 26: 1387.19, 16.00, 1403.19 52.00 MiB
169
+ Layer 27: 1387.19, 16.00, 1403.19 52.00 MiB
170
+ Layer 28: 1387.19, 16.00, 1403.19 52.00 MiB
171
+ Layer 29: 1387.19, 16.00, 1403.19 52.00 MiB
172
+ Layer 30: 1387.19, 16.00, 1403.19 52.00 MiB
173
+ Layer 31: 1387.19, 16.00, 1403.19 52.00 MiB
174
+ Layer 32: 1387.19, 16.00, 1403.19 52.00 MiB
175
+ Layer 33: 1387.19, 16.00, 1403.19 52.00 MiB
176
+ Layer 34: 1387.19, 16.00, 1403.19 52.00 MiB
177
+ Layer 35: 1387.19, 16.00, 1403.19 52.00 MiB
178
+ Layer 36: 1387.19, 16.00, 1403.19 52.00 MiB
179
+ Layer 37: 1387.19, 16.00, 1403.19 52.00 MiB
180
+ Layer 38: 1387.19, 16.00, 1403.19 52.00 MiB
181
+ Layer 39: 1387.19, 16.00, 1403.19 52.00 MiB
182
+ Layer 40: 1387.19, 16.00, 1403.19 52.00 MiB
183
+ Layer 41: 1387.19, 16.00, 1403.19 52.00 MiB
184
+ Layer 42: 1387.19, 16.00, 1403.19 52.00 MiB
185
+ Layer 43: 1387.19, 16.00, 1403.19 52.00 MiB
186
+ Layer 44: 1387.19, 16.00, 1403.19 52.00 MiB
187
+ Layer 45: 1387.19, 16.00, 1403.19 52.00 MiB
188
+ Layer 46: 1387.19, 16.00, 1403.19 52.00 MiB
189
+ Layer 47: 1387.19, 16.00, 1403.19 52.00 MiB
190
+ Layer 48: 1387.19, 16.00, 1403.19 52.00 MiB
191
+ Layer 49: 1387.19, 16.00, 1403.19 52.00 MiB
192
+ Layer 50: 1387.19, 16.00, 1403.19 52.00 MiB
193
+ Layer 51: 1387.19, 16.00, 1403.19 52.00 MiB
194
+ Layer 52: 1387.19, 16.00, 1403.19 52.00 MiB
195
+ Layer 53: 1387.19, 16.00, 1403.19 52.00 MiB
196
+ Layer 54: 1387.19, 16.00, 1403.19 52.00 MiB
197
+ Layer 55: 1387.19, 16.00, 1403.19 52.00 MiB
198
+ Layer 56: 1387.19, 16.00, 1403.19 52.00 MiB
199
+ Layer 57: 1387.19, 16.00, 1403.19 52.00 MiB
200
+ Layer 58: 1387.19, 16.00, 1403.19 52.00 MiB
201
+ Layer 59: 1387.19, 16.00, 1403.19 52.00 MiB
202
+ Layer 60: 1387.19, 16.00, 1403.19 52.00 MiB
203
+ Layer 61: 1387.19, 16.00, 1403.19 52.00 MiB
204
+ Layer 62: 1387.19, 16.00, 1403.19 52.00 MiB
205
+ Layer 63: 1387.19, 16.00, 1403.19 52.00 MiB
206
+ Layer 64: 1387.19, 16.00, 1403.19 52.00 MiB
207
+ Layer 65: 1387.19, 16.00, 1403.19 52.00 MiB
208
+ Layer 66: 1387.19, 16.00, 1403.19 52.00 MiB
209
+ Layer 67: 1387.19, 16.00, 1403.19 52.00 MiB
210
+ Layer 68: 1387.19, 16.00, 1403.19 52.00 MiB
211
+ Layer 69: 1387.19, 16.00, 1403.19 52.00 MiB
212
+ Layer 70: 321.56, 0.00, 321.56 MiB (output layer)
213
+ --------------------------------------------------------------------------
214
+ Total : 93363.29, 1120.00, 94483.29 MiB
215
+ Memory required for model tensors + cache: 94805 MiB
216
+ Memory available on all devices - compute: 95379 MiB
217
+ llm_load_tensors: ggml ctx size = 1.02 MiB
218
+ llm_load_tensors: offloading 70 repeating layers to GPU
219
+ llm_load_tensors: offloading non-repeating layers to GPU
220
+ llm_load_tensors: offloaded 71/71 layers to GPU
221
+ llm_load_tensors: CPU buffer size = 321.56 MiB
222
+ llm_load_tensors: CUDA0 buffer size = 93684.87 MiB
223
+ ....................................................................................................
224
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
225
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
226
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
227
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
228
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
229
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
230
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
231
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
232
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
233
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
234
+ llm_load_tensors: dense parameters loaded in 110.62s (5.44 GiB), expert parameters deferred (86.37 GiB)
235
+ llama_init_from_model: n_ctx = 4096
236
+ llama_init_from_model: n_batch = 512
237
+ llama_init_from_model: n_ubatch = 128
238
+ llama_init_from_model: flash_attn = 1
239
+ llama_init_from_model: attn_max_b = 0
240
+ llama_init_from_model: fused_moe = 1
241
+ llama_init_from_model: grouped er = 0
242
+ llama_init_from_model: fused_up_gate = 1
243
+ llama_init_from_model: fused_mmad = 1
244
+ llama_init_from_model: rope_cache = 0
245
+ llama_init_from_model: graph_reuse = 1
246
+ llama_init_from_model: k_cache_hadam = 0
247
+ llama_init_from_model: v_cache_hadam = 0
248
+ llama_init_from_model: split_mode_graph_scheduling = 0
249
+ llama_init_from_model: reduce_type = f16
250
+ llama_init_from_model: sched_async = 0
251
+ llama_init_from_model: ser = -1, 0
252
+ llama_init_from_model: freq_base = 500000.0
253
+ llama_init_from_model: freq_scale = 0.015625
254
+ llama_kv_cache_init: CUDA0 KV buffer size = 1120.00 MiB
255
+ llama_init_from_model: KV self size = 1120.00 MiB, K (f16): 560.00 MiB, V (f16): 560.00 MiB
256
+ llama_init_from_model: CUDA_Host output buffer size = 0.38 MiB
257
+ llama_init_from_model: CUDA0 compute buffer size = 51.00 MiB
258
+ llama_init_from_model: CUDA_Host compute buffer size = 3.00 MiB
259
+ llama_init_from_model: graph nodes = 3174
260
+ llama_init_from_model: graph splits = 2
261
+ llama_init_from_model: enabling only_active_experts scheduling
262
+
263
+ system_info: n_threads = 20 / 20 | AVX = 0 | AVX_VNNI = 0 | AVX2 = 0 | AVX512 = 0 | AVX512_VBMI = 0 | AVX512_VNNI = 0 | AVX512_BF16 = 0 | FMA = 0 | NEON = 1 | SVE = 0 | ARM_FMA = 1 | F16C = 0 | FP16_VA = 1 | WASM_SIMD = 0 | SSE3 = 0 | SSSE3 = 0 | VSX = 0 | MATMUL_INT8 = 0 |
264
+ perplexity: tokenizing the input ..
265
+ perplexity: tokenization took 1789.63 ms
266
+ perplexity: calculating perplexity over 8 chunks, n_ctx=4096, batch_size=512, n_seq=1
267
+ perplexity: 25.98 seconds per pass - ETA 3.45 minutes
268
+ [1]1.3713,[2]2.2535,[3]1.8046,[4]2.4539,[5]2.6550,[6]2.6851,[7]2.6250,[8]2.4683,
269
+ llama_print_timings: load time = 128307.95 ms
270
+ llama_print_timings: sample time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
271
+ llama_print_timings: prompt eval time = 209390.13 ms / 32768 tokens ( 6.39 ms per token, 156.49 tokens per second)
272
+ llama_print_timings: eval time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
273
+ llama_print_timings: total time = 322931.50 ms / 32769 tokens
274
+ ~ggml_backend_cuda_context: have 1 graphs
275
+
276
+ Final estimate: PPL over 8 chunks for n_ctx=4096 = 2.4683 +/- 0.04051
IQ3_S/quantize-command.txt ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ quant=IQ3_S
2
+ lane=vanilla-core
3
+ started_at=2026-06-21T01:22:11-04:00
4
+ bf16_first=/home/janet/laguna-gguf/laguna-m1/bf16/Laguna-M.1-BF16-00001-of-00010.gguf
5
+ imatrix=/mnt/pool/gguf/laguna-m1/imatrix/q8-public-v1-256-20260620-231949/laguna-m1-q8-public-v1-256-gatefix.imatrix
6
+ corpus=/mnt/pool/gguf/laguna-m1/calib/laguna-m1-imatrix-calibration-v1.txt
7
+ threads=20
8
+ command=/home/janet/src/ik_llama.cpp/build-cuda/bin/llama-quantize --keep-split --partial-requant --imatrix /mnt/pool/gguf/laguna-m1/imatrix/q8-public-v1-256-20260620-231949/laguna-m1-q8-public-v1-256-gatefix.imatrix --output-tensor-type q6_K --token-embedding-type q6_K /home/janet/laguna-gguf/laguna-m1/bf16/Laguna-M.1-BF16-00001-of-00010.gguf /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1.gguf IQ3_S 20
IQ3_S/quantize.log ADDED
The diff for this file is too large to render. See raw diff
 
IQ3_S/result.tsv ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ quant lane status files bytes sha256sums ppl ppl_unc kld kld_unc kl_reference completed_at log
2
+ IQ3_S vanilla-core finished 10 98576647840 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/SHA256SUMS 2.4683 0.04051 pending pending pending 2026-06-21T02:40:21-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/quantize.log
IQ3_S/smoke.log ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Log start
2
+ main: build = 4641 (be7d53ce)
3
+ main: built with cc (Ubuntu 13.3.0-6ubuntu2~24.04.1) 13.3.0 for aarch64-linux-gnu
4
+ main: seed = 123
5
+ ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
6
+ ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
7
+ ggml_cuda_init: found 1 CUDA devices:
8
+ Device 0: NVIDIA GB10, compute capability 12.1, VMM: yes, VRAM: 124610 MiB
9
+ CUDA0: using device CUDA0 - 12326 MiB free
10
+ llama_model_loader: additional 9 GGUFs metadata loaded.
11
+ llama_model_loader: loaded meta data with 59 key-value pairs and 1178 tensors from /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf (version GGUF V3 (latest))
12
+ llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
13
+ llama_model_loader: - kv 0: general.architecture str = laguna
14
+ llama_model_loader: - kv 1: general.type str = model
15
+ llama_model_loader: - kv 2: general.name str = Laguna M.1
16
+ llama_model_loader: - kv 3: general.size_label str = 256x11B
17
+ llama_model_loader: - kv 4: general.license str = apache-2.0
18
+ llama_model_loader: - kv 5: general.tags arr[str,6] = ["laguna-m.1", "vllm", "sglang", "bf1...
19
+ llama_model_loader: - kv 6: laguna.context_length u32 = 262144
20
+ llama_model_loader: - kv 7: laguna.embedding_length u32 = 4096
21
+ llama_model_loader: - kv 8: laguna.block_count u32 = 70
22
+ llama_model_loader: - kv 9: laguna.feed_forward_length u32 = 16384
23
+ llama_model_loader: - kv 10: laguna.attention.head_count arr[i32,70] = [64, 64, 64, 64, 64, 64, 64, 64, 64, ...
24
+ llama_model_loader: - kv 11: laguna.attention.head_count_kv u32 = 8
25
+ llama_model_loader: - kv 12: laguna.attention.key_length u32 = 128
26
+ llama_model_loader: - kv 13: laguna.attention.value_length u32 = 128
27
+ llama_model_loader: - kv 14: laguna.attention.layer_norm_rms_epsilon f32 = 0.000001
28
+ llama_model_loader: - kv 15: general.file_type u32 = 26
29
+ llama_model_loader: - kv 16: laguna.attention.sliding_window u32 = 0
30
+ llama_model_loader: - kv 17: laguna.rope.dimension_count u32 = 128
31
+ llama_model_loader: - kv 18: laguna.rope.dimension_count_swa u32 = 128
32
+ llama_model_loader: - kv 19: laguna.rope.freq_base f32 = 500000.000000
33
+ llama_model_loader: - kv 20: laguna.rope.freq_base_swa f32 = 10000.000000
34
+ llama_model_loader: - kv 21: laguna.rope.scaling.type str = yarn
35
+ llama_model_loader: - kv 22: laguna.rope.scaling.factor f32 = 64.000000
36
+ llama_model_loader: - kv 23: laguna.rope.scaling.original_context_length u32 = 4096
37
+ llama_model_loader: - kv 24: laguna.rope.scaling.yarn_ext_factor f32 = 1.000000
38
+ llama_model_loader: - kv 25: laguna.rope.scaling.yarn_attn_factor f32 = 1.000000
39
+ llama_model_loader: - kv 26: laguna.rope.scaling.yarn_beta_fast f32 = 64.000000
40
+ llama_model_loader: - kv 27: laguna.rope.scaling.yarn_beta_slow f32 = 1.000000
41
+ llama_model_loader: - kv 28: laguna.expert_count u32 = 256
42
+ llama_model_loader: - kv 29: laguna.expert_used_count u32 = 16
43
+ llama_model_loader: - kv 30: laguna.expert_feed_forward_length u32 = 1024
44
+ llama_model_loader: - kv 31: laguna.expert_shared_feed_forward_length u32 = 1024
45
+ llama_model_loader: - kv 32: laguna.expert_weights_scale f32 = 1.000000
46
+ llama_model_loader: - kv 33: laguna.expert_weights_norm bool = true
47
+ llama_model_loader: - kv 34: laguna.expert_gating_func u32 = 2
48
+ llama_model_loader: - kv 35: laguna.leading_dense_block_count u32 = 3
49
+ llama_model_loader: - kv 36: tokenizer.ggml.model str = gpt2
50
+ llama_model_loader: - kv 37: tokenizer.ggml.pre str = laguna
51
+ llama_model_loader: - kv 38: tokenizer.ggml.tokens arr[str,100352] = ["γ€ˆ|UNK|〉", "γ€ˆ|CODE_START|〉",...
52
+ llama_model_loader: - kv 39: tokenizer.ggml.token_type arr[i32,100352] = [3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, ...
53
+ llama_model_loader: - kv 40: tokenizer.ggml.merges arr[str,100026] = ["i n", "Δ  t", "Δ  Δ ", "e r", "Δ  a...
54
+ llama_model_loader: - kv 41: tokenizer.ggml.bos_token_id u32 = 2
55
+ llama_model_loader: - kv 42: tokenizer.ggml.eos_token_id u32 = 2
56
+ llama_model_loader: - kv 43: tokenizer.ggml.unknown_token_id u32 = 0
57
+ llama_model_loader: - kv 44: tokenizer.ggml.seperator_token_id u32 = 8
58
+ llama_model_loader: - kv 45: tokenizer.ggml.padding_token_id u32 = 9
59
+ llama_model_loader: - kv 46: tokenizer.ggml.mask_token_id u32 = 12
60
+ llama_model_loader: - kv 47: tokenizer.ggml.add_bos_token bool = true
61
+ llama_model_loader: - kv 48: tokenizer.ggml.add_sep_token bool = true
62
+ llama_model_loader: - kv 49: tokenizer.chat_template str = {#- Copied from laguna_glm_thinking_v...
63
+ llama_model_loader: - kv 50: tokenizer.ggml.eot_token_id u32 = 24
64
+ llama_model_loader: - kv 51: general.quantization_version u32 = 2
65
+ llama_model_loader: - kv 52: quantize.imatrix.file str = /mnt/pool/gguf/laguna-m1/imatrix/q8-p...
66
+ llama_model_loader: - kv 53: quantize.imatrix.dataset str = /workspace/laguna-m1/corpus/laguna-m1...
67
+ llama_model_loader: - kv 54: quantize.imatrix.entries_count i32 = 828
68
+ llama_model_loader: - kv 55: quantize.imatrix.chunks_count i32 = 2048
69
+ llama_model_loader: - kv 56: split.no u16 = 0
70
+ llama_model_loader: - kv 57: split.count u16 = 10
71
+ llama_model_loader: - kv 58: split.tensors.count i32 = 1178
72
+ llama_model_loader: - type f32: 415 tensors
73
+ llama_model_loader: - type q8_0: 140 tensors
74
+ llama_model_loader: - type q5_K: 70 tensors
75
+ llama_model_loader: - type q6_K: 2 tensors
76
+ llama_model_loader: - type iq3_s: 551 tensors
77
+ load: 0 unused tokens
78
+ load: special_eos_id is not in special_eog_ids - the tokenizer config may be incorrect
79
+ load: special_eot_id is not in special_eog_ids - the tokenizer config may be incorrect
80
+ load: printing all EOG tokens:
81
+ load: - 2 ('γ€ˆ|EOS|〉')
82
+ load: - 24 ('</assistant>')
83
+ load: special tokens cache size = 70
84
+ load: token to piece cache size = 0.6432 MB
85
+ llm_load_print_meta: format = GGUF V3 (latest)
86
+ llm_load_print_meta: arch = laguna
87
+ llm_load_print_meta: n_ctx_train = 262144
88
+ llm_load_print_meta: n_embd = 4096
89
+ llm_load_print_meta: n_layer = 70
90
+ llm_load_print_meta: n_head = 64
91
+ llm_load_print_meta: n_head_kv = 8
92
+ llm_load_print_meta: n_rot = 128
93
+ llm_load_print_meta: n_swa = 0
94
+ llm_load_print_meta: n_swa_pattern = 1
95
+ llm_load_print_meta: n_embd_head_k = 128
96
+ llm_load_print_meta: n_embd_head_v = 128
97
+ llm_load_print_meta: n_gqa = 8
98
+ llm_load_print_meta: n_embd_k_gqa = 1024
99
+ llm_load_print_meta: n_embd_v_gqa = 1024
100
+ llm_load_print_meta: f_norm_eps = 0.0e+00
101
+ llm_load_print_meta: f_norm_rms_eps = 1.0e-06
102
+ llm_load_print_meta: f_clamp_kqv = 0.0e+00
103
+ llm_load_print_meta: f_max_alibi_bias = 0.0e+00
104
+ llm_load_print_meta: f_logit_scale = 0.0e+00
105
+ llm_load_print_meta: n_ff = 16384
106
+ llm_load_print_meta: n_expert = 256
107
+ llm_load_print_meta: n_expert_used = 16
108
+ llm_load_print_meta: causal attn = 1
109
+ llm_load_print_meta: pooling type = 0
110
+ llm_load_print_meta: rope type = 2
111
+ llm_load_print_meta: rope scaling = yarn
112
+ llm_load_print_meta: freq_base_train = 500000.0
113
+ llm_load_print_meta: freq_scale_train = 0.015625
114
+ llm_load_print_meta: n_ctx_orig_yarn = 4096
115
+ llm_load_print_meta: rope_finetuned = unknown
116
+ llm_load_print_meta: ssm_d_conv = 0
117
+ llm_load_print_meta: ssm_d_inner = 0
118
+ llm_load_print_meta: ssm_d_state = 0
119
+ llm_load_print_meta: ssm_dt_rank = 0
120
+ llm_load_print_meta: ssm_n_group = 0
121
+ llm_load_print_meta: model type = ?B
122
+ llm_load_print_meta: model ftype = IQ3_S - 3.4375 bpw
123
+ llm_load_print_meta: model params = 225.796 B
124
+ llm_load_print_meta: model size = 91.803 GiB (3.492 BPW)
125
+ llm_load_print_meta: repeating layers = 91.175 GiB (3.481 BPW, 224.974 B parameters)
126
+ llm_load_print_meta: general.name = Laguna M.1
127
+ print_info: vocab type = BPE
128
+ print_info: n_vocab = 100352
129
+ print_info: n_merges = 100026
130
+ print_info: BOS token = 2 'γ€ˆ|EOS|〉'
131
+ print_info: EOS token = 2 'γ€ˆ|EOS|〉'
132
+ print_info: EOT token = 24 '</assistant>'
133
+ print_info: UNK token = 0 'γ€ˆ|UNK|〉'
134
+ print_info: SEP token = 8 'γ€ˆ|SEP|〉'
135
+ print_info: PAD token = 9 'γ€ˆ|PAD|〉'
136
+ print_info: MASK token = 12 'γ€ˆ|MASK|〉'
137
+ print_info: LF token = 268 'Ċ'
138
+ print_info: EOG token = 2 'γ€ˆ|EOS|〉'
139
+ print_info: EOG token = 24 '</assistant>'
140
+ print_info: max token length = 830
141
+ ======================================= HAVE_FANCY_SIMD is NOT defined
142
+ ------------------- Layer sizes:
143
+ Layer 0: 140.53, 8.00, 148.53 22.00 MiB
144
+ Layer 1: 140.53, 8.00, 148.53 22.00 MiB
145
+ Layer 2: 140.53, 8.00, 148.53 22.00 MiB
146
+ Layer 3: 1387.19, 8.00, 1395.19 52.00 MiB
147
+ Layer 4: 1387.19, 8.00, 1395.19 52.00 MiB
148
+ Layer 5: 1387.19, 8.00, 1395.19 52.00 MiB
149
+ Layer 6: 1387.19, 8.00, 1395.19 52.00 MiB
150
+ Layer 7: 1387.19, 8.00, 1395.19 52.00 MiB
151
+ Layer 8: 1387.19, 8.00, 1395.19 52.00 MiB
152
+ Layer 9: 1387.19, 8.00, 1395.19 52.00 MiB
153
+ Layer 10: 1387.19, 8.00, 1395.19 52.00 MiB
154
+ Layer 11: 1387.19, 8.00, 1395.19 52.00 MiB
155
+ Layer 12: 1387.19, 8.00, 1395.19 52.00 MiB
156
+ Layer 13: 1387.19, 8.00, 1395.19 52.00 MiB
157
+ Layer 14: 1387.19, 8.00, 1395.19 52.00 MiB
158
+ Layer 15: 1387.19, 8.00, 1395.19 52.00 MiB
159
+ Layer 16: 1387.19, 8.00, 1395.19 52.00 MiB
160
+ Layer 17: 1387.19, 8.00, 1395.19 52.00 MiB
161
+ Layer 18: 1387.19, 8.00, 1395.19 52.00 MiB
162
+ Layer 19: 1387.19, 8.00, 1395.19 52.00 MiB
163
+ Layer 20: 1387.19, 8.00, 1395.19 52.00 MiB
164
+ Layer 21: 1387.19, 8.00, 1395.19 52.00 MiB
165
+ Layer 22: 1387.19, 8.00, 1395.19 52.00 MiB
166
+ Layer 23: 1387.19, 8.00, 1395.19 52.00 MiB
167
+ Layer 24: 1387.19, 8.00, 1395.19 52.00 MiB
168
+ Layer 25: 1387.19, 8.00, 1395.19 52.00 MiB
169
+ Layer 26: 1387.19, 8.00, 1395.19 52.00 MiB
170
+ Layer 27: 1387.19, 8.00, 1395.19 52.00 MiB
171
+ Layer 28: 1387.19, 8.00, 1395.19 52.00 MiB
172
+ Layer 29: 1387.19, 8.00, 1395.19 52.00 MiB
173
+ Layer 30: 1387.19, 8.00, 1395.19 52.00 MiB
174
+ Layer 31: 1387.19, 8.00, 1395.19 52.00 MiB
175
+ Layer 32: 1387.19, 8.00, 1395.19 52.00 MiB
176
+ Layer 33: 1387.19, 8.00, 1395.19 52.00 MiB
177
+ Layer 34: 1387.19, 8.00, 1395.19 52.00 MiB
178
+ Layer 35: 1387.19, 8.00, 1395.19 52.00 MiB
179
+ Layer 36: 1387.19, 8.00, 1395.19 52.00 MiB
180
+ Layer 37: 1387.19, 8.00, 1395.19 52.00 MiB
181
+ Layer 38: 1387.19, 8.00, 1395.19 52.00 MiB
182
+ Layer 39: 1387.19, 8.00, 1395.19 52.00 MiB
183
+ Layer 40: 1387.19, 8.00, 1395.19 52.00 MiB
184
+ Layer 41: 1387.19, 8.00, 1395.19 52.00 MiB
185
+ Layer 42: 1387.19, 8.00, 1395.19 52.00 MiB
186
+ Layer 43: 1387.19, 8.00, 1395.19 52.00 MiB
187
+ Layer 44: 1387.19, 8.00, 1395.19 52.00 MiB
188
+ Layer 45: 1387.19, 8.00, 1395.19 52.00 MiB
189
+ Layer 46: 1387.19, 8.00, 1395.19 52.00 MiB
190
+ Layer 47: 1387.19, 8.00, 1395.19 52.00 MiB
191
+ Layer 48: 1387.19, 8.00, 1395.19 52.00 MiB
192
+ Layer 49: 1387.19, 8.00, 1395.19 52.00 MiB
193
+ Layer 50: 1387.19, 8.00, 1395.19 52.00 MiB
194
+ Layer 51: 1387.19, 8.00, 1395.19 52.00 MiB
195
+ Layer 52: 1387.19, 8.00, 1395.19 52.00 MiB
196
+ Layer 53: 1387.19, 8.00, 1395.19 52.00 MiB
197
+ Layer 54: 1387.19, 8.00, 1395.19 52.00 MiB
198
+ Layer 55: 1387.19, 8.00, 1395.19 52.00 MiB
199
+ Layer 56: 1387.19, 8.00, 1395.19 52.00 MiB
200
+ Layer 57: 1387.19, 8.00, 1395.19 52.00 MiB
201
+ Layer 58: 1387.19, 8.00, 1395.19 52.00 MiB
202
+ Layer 59: 1387.19, 8.00, 1395.19 52.00 MiB
203
+ Layer 60: 1387.19, 8.00, 1395.19 52.00 MiB
204
+ Layer 61: 1387.19, 8.00, 1395.19 52.00 MiB
205
+ Layer 62: 1387.19, 8.00, 1395.19 52.00 MiB
206
+ Layer 63: 1387.19, 8.00, 1395.19 52.00 MiB
207
+ Layer 64: 1387.19, 8.00, 1395.19 52.00 MiB
208
+ Layer 65: 1387.19, 8.00, 1395.19 52.00 MiB
209
+ Layer 66: 1387.19, 8.00, 1395.19 52.00 MiB
210
+ Layer 67: 1387.19, 8.00, 1395.19 52.00 MiB
211
+ Layer 68: 1387.19, 8.00, 1395.19 52.00 MiB
212
+ Layer 69: 1387.19, 8.00, 1395.19 52.00 MiB
213
+ Layer 70: 321.56, 0.00, 321.56 MiB (output layer)
214
+ --------------------------------------------------------------------------
215
+ Total : 93363.29, 560.00, 93923.29 MiB
216
+ Memory required for model tensors + cache: 94245 MiB
217
+ Memory available on all devices - compute: 11206 MiB
218
+ llm_load_tensors: ggml ctx size = 1.02 MiB
219
+ llm_load_tensors: offloading 70 repeating layers to GPU
220
+ llm_load_tensors: offloading non-repeating layers to GPU
221
+ llm_load_tensors: offloaded 71/71 layers to GPU
222
+ llm_load_tensors: CPU buffer size = 321.56 MiB
223
+ llm_load_tensors: CUDA0 buffer size = 93684.87 MiB
224
+ ....................................................................................................
225
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
226
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
227
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
228
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
229
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
230
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
231
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
232
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
233
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
234
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
235
+ llm_load_tensors: dense parameters loaded in 125.48s (5.44 GiB), expert parameters deferred (86.37 GiB)
236
+ llama_init_from_model: n_ctx = 2048
237
+ llama_init_from_model: n_batch = 256
238
+ llama_init_from_model: n_ubatch = 128
239
+ llama_init_from_model: flash_attn = 1
240
+ llama_init_from_model: attn_max_b = 0
241
+ llama_init_from_model: fused_moe = 1
242
+ llama_init_from_model: grouped er = 0
243
+ llama_init_from_model: fused_up_gate = 1
244
+ llama_init_from_model: fused_mmad = 1
245
+ llama_init_from_model: rope_cache = 0
246
+ llama_init_from_model: graph_reuse = 1
247
+ llama_init_from_model: k_cache_hadam = 0
248
+ llama_init_from_model: v_cache_hadam = 0
249
+ llama_init_from_model: split_mode_graph_scheduling = 0
250
+ llama_init_from_model: reduce_type = f16
251
+ llama_init_from_model: sched_async = 0
252
+ llama_init_from_model: ser = -1, 0
253
+ llama_init_from_model: freq_base = 500000.0
254
+ llama_init_from_model: freq_scale = 0.015625
255
+ llama_kv_cache_init: CUDA0 KV buffer size = 560.00 MiB
256
+ llama_init_from_model: KV self size = 560.00 MiB, K (f16): 280.00 MiB, V (f16): 280.00 MiB
257
+ llama_init_from_model: CUDA_Host output buffer size = 0.38 MiB
258
+ llama_init_from_model: CUDA0 compute buffer size = 51.00 MiB
259
+ llama_init_from_model: CUDA_Host compute buffer size = 2.50 MiB
260
+ llama_init_from_model: graph nodes = 3174
261
+ llama_init_from_model: graph splits = 2
262
+ llama_init_from_model: enabling only_active_experts scheduling
263
+
264
+ system_info: n_threads = 20 / 20 | AVX = 0 | AVX_VNNI = 0 | AVX2 = 0 | AVX512 = 0 | AVX512_VBMI = 0 | AVX512_VNNI = 0 | AVX512_BF16 = 0 | FMA = 0 | NEON = 1 | SVE = 0 | ARM_FMA = 1 | F16C = 0 | FP16_VA = 1 | WASM_SIMD = 0 | SSE3 = 0 | SSSE3 = 0 | VSX = 0 | MATMUL_INT8 = 0 |
265
+ sampling:
266
+ repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000
267
+ top_k = 40, tfs_z = 1.000, top_p = 0.950, min_p = 0.050, typical_p = 1.000, temp = 0.200
268
+ mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000
269
+ xtc_probability = 0.000, xtc_threshold = 1.000, top_n_sigma = 0.000
270
+ adaptive_target = -1.00, adaptive_decay = 0.90
271
+ sampling order:
272
+ CFG -> Penalties -> dry -> top_k -> tfs_z -> typical_p -> top_p -> min_p -> xtc -> top_n_sigma -> temperature -> adaptive_p
273
+ generate: n_ctx = 2048, n_batch = 256, n_predict = 48, n_keep = 1
274
+
275
+
276
+ Write a tiny Python function that computes Fibonacci numbers iteratively. Your function should be named 'fibonacci' and take a single integer argument 'n'. It should return the nth Fibonacci number. The function should handle edge cases where n is 0 or 1.
277
+
278
+ We are computing the Fibonacci sequence iteratively
279
+ llama_print_timings: load time = 126743.48 ms
280
+ llama_print_timings: sample time = 1.94 ms / 48 runs ( 0.04 ms per token, 24742.27 tokens per second)
281
+ llama_print_timings: prompt eval time = 952.45 ms / 12 tokens ( 79.37 ms per token, 12.60 tokens per second)
282
+ llama_print_timings: eval time = 3427.81 ms / 47 runs ( 72.93 ms per token, 13.71 tokens per second)
283
+ llama_print_timings: total time = 130246.48 ms / 59 tokens
284
+ ~ggml_backend_cuda_context: have 2 graphs
285
+ Log end
IQ3_S/smoke.ngl99.log ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Log start
2
+ main: build = 4641 (be7d53ce)
3
+ main: built with cc (Ubuntu 13.3.0-6ubuntu2~24.04.1) 13.3.0 for aarch64-linux-gnu
4
+ main: seed = 123
5
+ ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
6
+ ggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no
7
+ ggml_cuda_init: found 1 CUDA devices:
8
+ Device 0: NVIDIA GB10, compute capability 12.1, VMM: yes, VRAM: 124610 MiB
9
+ CUDA0: using device CUDA0 - 12326 MiB free
10
+ llama_model_loader: additional 9 GGUFs metadata loaded.
11
+ llama_model_loader: loaded meta data with 59 key-value pairs and 1178 tensors from /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf (version GGUF V3 (latest))
12
+ llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
13
+ llama_model_loader: - kv 0: general.architecture str = laguna
14
+ llama_model_loader: - kv 1: general.type str = model
15
+ llama_model_loader: - kv 2: general.name str = Laguna M.1
16
+ llama_model_loader: - kv 3: general.size_label str = 256x11B
17
+ llama_model_loader: - kv 4: general.license str = apache-2.0
18
+ llama_model_loader: - kv 5: general.tags arr[str,6] = ["laguna-m.1", "vllm", "sglang", "bf1...
19
+ llama_model_loader: - kv 6: laguna.context_length u32 = 262144
20
+ llama_model_loader: - kv 7: laguna.embedding_length u32 = 4096
21
+ llama_model_loader: - kv 8: laguna.block_count u32 = 70
22
+ llama_model_loader: - kv 9: laguna.feed_forward_length u32 = 16384
23
+ llama_model_loader: - kv 10: laguna.attention.head_count arr[i32,70] = [64, 64, 64, 64, 64, 64, 64, 64, 64, ...
24
+ llama_model_loader: - kv 11: laguna.attention.head_count_kv u32 = 8
25
+ llama_model_loader: - kv 12: laguna.attention.key_length u32 = 128
26
+ llama_model_loader: - kv 13: laguna.attention.value_length u32 = 128
27
+ llama_model_loader: - kv 14: laguna.attention.layer_norm_rms_epsilon f32 = 0.000001
28
+ llama_model_loader: - kv 15: general.file_type u32 = 26
29
+ llama_model_loader: - kv 16: laguna.attention.sliding_window u32 = 0
30
+ llama_model_loader: - kv 17: laguna.rope.dimension_count u32 = 128
31
+ llama_model_loader: - kv 18: laguna.rope.dimension_count_swa u32 = 128
32
+ llama_model_loader: - kv 19: laguna.rope.freq_base f32 = 500000.000000
33
+ llama_model_loader: - kv 20: laguna.rope.freq_base_swa f32 = 10000.000000
34
+ llama_model_loader: - kv 21: laguna.rope.scaling.type str = yarn
35
+ llama_model_loader: - kv 22: laguna.rope.scaling.factor f32 = 64.000000
36
+ llama_model_loader: - kv 23: laguna.rope.scaling.original_context_length u32 = 4096
37
+ llama_model_loader: - kv 24: laguna.rope.scaling.yarn_ext_factor f32 = 1.000000
38
+ llama_model_loader: - kv 25: laguna.rope.scaling.yarn_attn_factor f32 = 1.000000
39
+ llama_model_loader: - kv 26: laguna.rope.scaling.yarn_beta_fast f32 = 64.000000
40
+ llama_model_loader: - kv 27: laguna.rope.scaling.yarn_beta_slow f32 = 1.000000
41
+ llama_model_loader: - kv 28: laguna.expert_count u32 = 256
42
+ llama_model_loader: - kv 29: laguna.expert_used_count u32 = 16
43
+ llama_model_loader: - kv 30: laguna.expert_feed_forward_length u32 = 1024
44
+ llama_model_loader: - kv 31: laguna.expert_shared_feed_forward_length u32 = 1024
45
+ llama_model_loader: - kv 32: laguna.expert_weights_scale f32 = 1.000000
46
+ llama_model_loader: - kv 33: laguna.expert_weights_norm bool = true
47
+ llama_model_loader: - kv 34: laguna.expert_gating_func u32 = 2
48
+ llama_model_loader: - kv 35: laguna.leading_dense_block_count u32 = 3
49
+ llama_model_loader: - kv 36: tokenizer.ggml.model str = gpt2
50
+ llama_model_loader: - kv 37: tokenizer.ggml.pre str = laguna
51
+ llama_model_loader: - kv 38: tokenizer.ggml.tokens arr[str,100352] = ["γ€ˆ|UNK|〉", "γ€ˆ|CODE_START|〉",...
52
+ llama_model_loader: - kv 39: tokenizer.ggml.token_type arr[i32,100352] = [3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, ...
53
+ llama_model_loader: - kv 40: tokenizer.ggml.merges arr[str,100026] = ["i n", "Δ  t", "Δ  Δ ", "e r", "Δ  a...
54
+ llama_model_loader: - kv 41: tokenizer.ggml.bos_token_id u32 = 2
55
+ llama_model_loader: - kv 42: tokenizer.ggml.eos_token_id u32 = 2
56
+ llama_model_loader: - kv 43: tokenizer.ggml.unknown_token_id u32 = 0
57
+ llama_model_loader: - kv 44: tokenizer.ggml.seperator_token_id u32 = 8
58
+ llama_model_loader: - kv 45: tokenizer.ggml.padding_token_id u32 = 9
59
+ llama_model_loader: - kv 46: tokenizer.ggml.mask_token_id u32 = 12
60
+ llama_model_loader: - kv 47: tokenizer.ggml.add_bos_token bool = true
61
+ llama_model_loader: - kv 48: tokenizer.ggml.add_sep_token bool = true
62
+ llama_model_loader: - kv 49: tokenizer.chat_template str = {#- Copied from laguna_glm_thinking_v...
63
+ llama_model_loader: - kv 50: tokenizer.ggml.eot_token_id u32 = 24
64
+ llama_model_loader: - kv 51: general.quantization_version u32 = 2
65
+ llama_model_loader: - kv 52: quantize.imatrix.file str = /mnt/pool/gguf/laguna-m1/imatrix/q8-p...
66
+ llama_model_loader: - kv 53: quantize.imatrix.dataset str = /workspace/laguna-m1/corpus/laguna-m1...
67
+ llama_model_loader: - kv 54: quantize.imatrix.entries_count i32 = 828
68
+ llama_model_loader: - kv 55: quantize.imatrix.chunks_count i32 = 2048
69
+ llama_model_loader: - kv 56: split.no u16 = 0
70
+ llama_model_loader: - kv 57: split.count u16 = 10
71
+ llama_model_loader: - kv 58: split.tensors.count i32 = 1178
72
+ llama_model_loader: - type f32: 415 tensors
73
+ llama_model_loader: - type q8_0: 140 tensors
74
+ llama_model_loader: - type q5_K: 70 tensors
75
+ llama_model_loader: - type q6_K: 2 tensors
76
+ llama_model_loader: - type iq3_s: 551 tensors
77
+ load: 0 unused tokens
78
+ load: special_eos_id is not in special_eog_ids - the tokenizer config may be incorrect
79
+ load: special_eot_id is not in special_eog_ids - the tokenizer config may be incorrect
80
+ load: printing all EOG tokens:
81
+ load: - 2 ('γ€ˆ|EOS|〉')
82
+ load: - 24 ('</assistant>')
83
+ load: special tokens cache size = 70
84
+ load: token to piece cache size = 0.6432 MB
85
+ llm_load_print_meta: format = GGUF V3 (latest)
86
+ llm_load_print_meta: arch = laguna
87
+ llm_load_print_meta: n_ctx_train = 262144
88
+ llm_load_print_meta: n_embd = 4096
89
+ llm_load_print_meta: n_layer = 70
90
+ llm_load_print_meta: n_head = 64
91
+ llm_load_print_meta: n_head_kv = 8
92
+ llm_load_print_meta: n_rot = 128
93
+ llm_load_print_meta: n_swa = 0
94
+ llm_load_print_meta: n_swa_pattern = 1
95
+ llm_load_print_meta: n_embd_head_k = 128
96
+ llm_load_print_meta: n_embd_head_v = 128
97
+ llm_load_print_meta: n_gqa = 8
98
+ llm_load_print_meta: n_embd_k_gqa = 1024
99
+ llm_load_print_meta: n_embd_v_gqa = 1024
100
+ llm_load_print_meta: f_norm_eps = 0.0e+00
101
+ llm_load_print_meta: f_norm_rms_eps = 1.0e-06
102
+ llm_load_print_meta: f_clamp_kqv = 0.0e+00
103
+ llm_load_print_meta: f_max_alibi_bias = 0.0e+00
104
+ llm_load_print_meta: f_logit_scale = 0.0e+00
105
+ llm_load_print_meta: n_ff = 16384
106
+ llm_load_print_meta: n_expert = 256
107
+ llm_load_print_meta: n_expert_used = 16
108
+ llm_load_print_meta: causal attn = 1
109
+ llm_load_print_meta: pooling type = 0
110
+ llm_load_print_meta: rope type = 2
111
+ llm_load_print_meta: rope scaling = yarn
112
+ llm_load_print_meta: freq_base_train = 500000.0
113
+ llm_load_print_meta: freq_scale_train = 0.015625
114
+ llm_load_print_meta: n_ctx_orig_yarn = 4096
115
+ llm_load_print_meta: rope_finetuned = unknown
116
+ llm_load_print_meta: ssm_d_conv = 0
117
+ llm_load_print_meta: ssm_d_inner = 0
118
+ llm_load_print_meta: ssm_d_state = 0
119
+ llm_load_print_meta: ssm_dt_rank = 0
120
+ llm_load_print_meta: ssm_n_group = 0
121
+ llm_load_print_meta: model type = ?B
122
+ llm_load_print_meta: model ftype = IQ3_S - 3.4375 bpw
123
+ llm_load_print_meta: model params = 225.796 B
124
+ llm_load_print_meta: model size = 91.803 GiB (3.492 BPW)
125
+ llm_load_print_meta: repeating layers = 91.175 GiB (3.481 BPW, 224.974 B parameters)
126
+ llm_load_print_meta: general.name = Laguna M.1
127
+ print_info: vocab type = BPE
128
+ print_info: n_vocab = 100352
129
+ print_info: n_merges = 100026
130
+ print_info: BOS token = 2 'γ€ˆ|EOS|〉'
131
+ print_info: EOS token = 2 'γ€ˆ|EOS|〉'
132
+ print_info: EOT token = 24 '</assistant>'
133
+ print_info: UNK token = 0 'γ€ˆ|UNK|〉'
134
+ print_info: SEP token = 8 'γ€ˆ|SEP|〉'
135
+ print_info: PAD token = 9 'γ€ˆ|PAD|〉'
136
+ print_info: MASK token = 12 'γ€ˆ|MASK|〉'
137
+ print_info: LF token = 268 'Ċ'
138
+ print_info: EOG token = 2 'γ€ˆ|EOS|〉'
139
+ print_info: EOG token = 24 '</assistant>'
140
+ print_info: max token length = 830
141
+ ======================================= HAVE_FANCY_SIMD is NOT defined
142
+ ------------------- Layer sizes:
143
+ Layer 0: 140.53, 8.00, 148.53 22.00 MiB
144
+ Layer 1: 140.53, 8.00, 148.53 22.00 MiB
145
+ Layer 2: 140.53, 8.00, 148.53 22.00 MiB
146
+ Layer 3: 1387.19, 8.00, 1395.19 52.00 MiB
147
+ Layer 4: 1387.19, 8.00, 1395.19 52.00 MiB
148
+ Layer 5: 1387.19, 8.00, 1395.19 52.00 MiB
149
+ Layer 6: 1387.19, 8.00, 1395.19 52.00 MiB
150
+ Layer 7: 1387.19, 8.00, 1395.19 52.00 MiB
151
+ Layer 8: 1387.19, 8.00, 1395.19 52.00 MiB
152
+ Layer 9: 1387.19, 8.00, 1395.19 52.00 MiB
153
+ Layer 10: 1387.19, 8.00, 1395.19 52.00 MiB
154
+ Layer 11: 1387.19, 8.00, 1395.19 52.00 MiB
155
+ Layer 12: 1387.19, 8.00, 1395.19 52.00 MiB
156
+ Layer 13: 1387.19, 8.00, 1395.19 52.00 MiB
157
+ Layer 14: 1387.19, 8.00, 1395.19 52.00 MiB
158
+ Layer 15: 1387.19, 8.00, 1395.19 52.00 MiB
159
+ Layer 16: 1387.19, 8.00, 1395.19 52.00 MiB
160
+ Layer 17: 1387.19, 8.00, 1395.19 52.00 MiB
161
+ Layer 18: 1387.19, 8.00, 1395.19 52.00 MiB
162
+ Layer 19: 1387.19, 8.00, 1395.19 52.00 MiB
163
+ Layer 20: 1387.19, 8.00, 1395.19 52.00 MiB
164
+ Layer 21: 1387.19, 8.00, 1395.19 52.00 MiB
165
+ Layer 22: 1387.19, 8.00, 1395.19 52.00 MiB
166
+ Layer 23: 1387.19, 8.00, 1395.19 52.00 MiB
167
+ Layer 24: 1387.19, 8.00, 1395.19 52.00 MiB
168
+ Layer 25: 1387.19, 8.00, 1395.19 52.00 MiB
169
+ Layer 26: 1387.19, 8.00, 1395.19 52.00 MiB
170
+ Layer 27: 1387.19, 8.00, 1395.19 52.00 MiB
171
+ Layer 28: 1387.19, 8.00, 1395.19 52.00 MiB
172
+ Layer 29: 1387.19, 8.00, 1395.19 52.00 MiB
173
+ Layer 30: 1387.19, 8.00, 1395.19 52.00 MiB
174
+ Layer 31: 1387.19, 8.00, 1395.19 52.00 MiB
175
+ Layer 32: 1387.19, 8.00, 1395.19 52.00 MiB
176
+ Layer 33: 1387.19, 8.00, 1395.19 52.00 MiB
177
+ Layer 34: 1387.19, 8.00, 1395.19 52.00 MiB
178
+ Layer 35: 1387.19, 8.00, 1395.19 52.00 MiB
179
+ Layer 36: 1387.19, 8.00, 1395.19 52.00 MiB
180
+ Layer 37: 1387.19, 8.00, 1395.19 52.00 MiB
181
+ Layer 38: 1387.19, 8.00, 1395.19 52.00 MiB
182
+ Layer 39: 1387.19, 8.00, 1395.19 52.00 MiB
183
+ Layer 40: 1387.19, 8.00, 1395.19 52.00 MiB
184
+ Layer 41: 1387.19, 8.00, 1395.19 52.00 MiB
185
+ Layer 42: 1387.19, 8.00, 1395.19 52.00 MiB
186
+ Layer 43: 1387.19, 8.00, 1395.19 52.00 MiB
187
+ Layer 44: 1387.19, 8.00, 1395.19 52.00 MiB
188
+ Layer 45: 1387.19, 8.00, 1395.19 52.00 MiB
189
+ Layer 46: 1387.19, 8.00, 1395.19 52.00 MiB
190
+ Layer 47: 1387.19, 8.00, 1395.19 52.00 MiB
191
+ Layer 48: 1387.19, 8.00, 1395.19 52.00 MiB
192
+ Layer 49: 1387.19, 8.00, 1395.19 52.00 MiB
193
+ Layer 50: 1387.19, 8.00, 1395.19 52.00 MiB
194
+ Layer 51: 1387.19, 8.00, 1395.19 52.00 MiB
195
+ Layer 52: 1387.19, 8.00, 1395.19 52.00 MiB
196
+ Layer 53: 1387.19, 8.00, 1395.19 52.00 MiB
197
+ Layer 54: 1387.19, 8.00, 1395.19 52.00 MiB
198
+ Layer 55: 1387.19, 8.00, 1395.19 52.00 MiB
199
+ Layer 56: 1387.19, 8.00, 1395.19 52.00 MiB
200
+ Layer 57: 1387.19, 8.00, 1395.19 52.00 MiB
201
+ Layer 58: 1387.19, 8.00, 1395.19 52.00 MiB
202
+ Layer 59: 1387.19, 8.00, 1395.19 52.00 MiB
203
+ Layer 60: 1387.19, 8.00, 1395.19 52.00 MiB
204
+ Layer 61: 1387.19, 8.00, 1395.19 52.00 MiB
205
+ Layer 62: 1387.19, 8.00, 1395.19 52.00 MiB
206
+ Layer 63: 1387.19, 8.00, 1395.19 52.00 MiB
207
+ Layer 64: 1387.19, 8.00, 1395.19 52.00 MiB
208
+ Layer 65: 1387.19, 8.00, 1395.19 52.00 MiB
209
+ Layer 66: 1387.19, 8.00, 1395.19 52.00 MiB
210
+ Layer 67: 1387.19, 8.00, 1395.19 52.00 MiB
211
+ Layer 68: 1387.19, 8.00, 1395.19 52.00 MiB
212
+ Layer 69: 1387.19, 8.00, 1395.19 52.00 MiB
213
+ Layer 70: 321.56, 0.00, 321.56 MiB (output layer)
214
+ --------------------------------------------------------------------------
215
+ Total : 93363.29, 560.00, 93923.29 MiB
216
+ Memory required for model tensors + cache: 94245 MiB
217
+ Memory available on all devices - compute: 11206 MiB
218
+ llm_load_tensors: ggml ctx size = 1.02 MiB
219
+ llm_load_tensors: offloading 70 repeating layers to GPU
220
+ llm_load_tensors: offloading non-repeating layers to GPU
221
+ llm_load_tensors: offloaded 71/71 layers to GPU
222
+ llm_load_tensors: CPU buffer size = 321.56 MiB
223
+ llm_load_tensors: CUDA0 buffer size = 93684.87 MiB
224
+ ....................................................................................................
225
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
226
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
227
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
228
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
229
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
230
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
231
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
232
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
233
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
234
+ warning: madvise(..., MADV_DONTNEED) failed: Cannot allocate memory
235
+ llm_load_tensors: dense parameters loaded in 125.48s (5.44 GiB), expert parameters deferred (86.37 GiB)
236
+ llama_init_from_model: n_ctx = 2048
237
+ llama_init_from_model: n_batch = 256
238
+ llama_init_from_model: n_ubatch = 128
239
+ llama_init_from_model: flash_attn = 1
240
+ llama_init_from_model: attn_max_b = 0
241
+ llama_init_from_model: fused_moe = 1
242
+ llama_init_from_model: grouped er = 0
243
+ llama_init_from_model: fused_up_gate = 1
244
+ llama_init_from_model: fused_mmad = 1
245
+ llama_init_from_model: rope_cache = 0
246
+ llama_init_from_model: graph_reuse = 1
247
+ llama_init_from_model: k_cache_hadam = 0
248
+ llama_init_from_model: v_cache_hadam = 0
249
+ llama_init_from_model: split_mode_graph_scheduling = 0
250
+ llama_init_from_model: reduce_type = f16
251
+ llama_init_from_model: sched_async = 0
252
+ llama_init_from_model: ser = -1, 0
253
+ llama_init_from_model: freq_base = 500000.0
254
+ llama_init_from_model: freq_scale = 0.015625
255
+ llama_kv_cache_init: CUDA0 KV buffer size = 560.00 MiB
256
+ llama_init_from_model: KV self size = 560.00 MiB, K (f16): 280.00 MiB, V (f16): 280.00 MiB
257
+ llama_init_from_model: CUDA_Host output buffer size = 0.38 MiB
258
+ llama_init_from_model: CUDA0 compute buffer size = 51.00 MiB
259
+ llama_init_from_model: CUDA_Host compute buffer size = 2.50 MiB
260
+ llama_init_from_model: graph nodes = 3174
261
+ llama_init_from_model: graph splits = 2
262
+ llama_init_from_model: enabling only_active_experts scheduling
263
+
264
+ system_info: n_threads = 20 / 20 | AVX = 0 | AVX_VNNI = 0 | AVX2 = 0 | AVX512 = 0 | AVX512_VBMI = 0 | AVX512_VNNI = 0 | AVX512_BF16 = 0 | FMA = 0 | NEON = 1 | SVE = 0 | ARM_FMA = 1 | F16C = 0 | FP16_VA = 1 | WASM_SIMD = 0 | SSE3 = 0 | SSSE3 = 0 | VSX = 0 | MATMUL_INT8 = 0 |
265
+ sampling:
266
+ repeat_last_n = 64, repeat_penalty = 1.000, frequency_penalty = 0.000, presence_penalty = 0.000
267
+ top_k = 40, tfs_z = 1.000, top_p = 0.950, min_p = 0.050, typical_p = 1.000, temp = 0.200
268
+ mirostat = 0, mirostat_lr = 0.100, mirostat_ent = 5.000
269
+ xtc_probability = 0.000, xtc_threshold = 1.000, top_n_sigma = 0.000
270
+ adaptive_target = -1.00, adaptive_decay = 0.90
271
+ sampling order:
272
+ CFG -> Penalties -> dry -> top_k -> tfs_z -> typical_p -> top_p -> min_p -> xtc -> top_n_sigma -> temperature -> adaptive_p
273
+ generate: n_ctx = 2048, n_batch = 256, n_predict = 48, n_keep = 1
274
+
275
+
276
+ Write a tiny Python function that computes Fibonacci numbers iteratively. Your function should be named 'fibonacci' and take a single integer argument 'n'. It should return the nth Fibonacci number. The function should handle edge cases where n is 0 or 1.
277
+
278
+ We are computing the Fibonacci sequence iteratively
279
+ llama_print_timings: load time = 126743.48 ms
280
+ llama_print_timings: sample time = 1.94 ms / 48 runs ( 0.04 ms per token, 24742.27 tokens per second)
281
+ llama_print_timings: prompt eval time = 952.45 ms / 12 tokens ( 79.37 ms per token, 12.60 tokens per second)
282
+ llama_print_timings: eval time = 3427.81 ms / 47 runs ( 72.93 ms per token, 13.71 tokens per second)
283
+ llama_print_timings: total time = 130246.48 ms / 59 tokens
284
+ ~ggml_backend_cuda_context: have 2 graphs
285
+ Log end
MANIFEST.tsv CHANGED
@@ -1,18 +1,21 @@
1
- IQ3_K_R4/eval-ppl.log 18274
2
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00001-of-00010.gguf 11377052512
3
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00002-of-00010.gguf 10150217440
4
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00003-of-00010.gguf 10150217440
5
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00004-of-00010.gguf 10150217440
6
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00005-of-00010.gguf 10150217440
7
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00006-of-00010.gguf 10150217440
8
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00007-of-00010.gguf 10150217440
9
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00008-of-00010.gguf 10150217440
10
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00009-of-00010.gguf 10150217440
11
- IQ3_K_R4/Laguna-M.1-IQ3_K_R4-imatrix-public-v1-00010-of-00010.gguf 5997855808
12
- IQ3_K_R4/quantize-command.txt 812
13
- IQ3_K_R4/quantize.log 174519
14
- IQ3_K_R4/result.tsv 351
15
- IQ3_K_R4/SHA256SUMS 1240
16
- IQ3_K_R4/smoke.log 18960
17
- README.md 5616
18
- results.tsv 1095
 
 
 
 
1
+ IQ3_S/eval-offload.tsv 192
2
+ IQ3_S/eval-ppl.log 18261
3
+ IQ3_S/eval-ppl.ngl99.log 18261
4
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00001-of-00010.gguf 11377052512
5
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00002-of-00010.gguf 10150217440
6
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00003-of-00010.gguf 10150217440
7
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00004-of-00010.gguf 10150217440
8
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00005-of-00010.gguf 10150217440
9
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00006-of-00010.gguf 10150217440
10
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00007-of-00010.gguf 10150217440
11
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00008-of-00010.gguf 10150217440
12
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00009-of-00010.gguf 10150217440
13
+ IQ3_S/Laguna-M.1-IQ3_S-imatrix-public-v1-00010-of-00010.gguf 5997855808
14
+ IQ3_S/quantize-command.txt 808
15
+ IQ3_S/quantize.log 172866
16
+ IQ3_S/result.tsv 341
17
+ IQ3_S/SHA256SUMS 1210
18
+ IQ3_S/smoke.log 18802
19
+ IQ3_S/smoke.ngl99.log 18802
20
+ README.md 5631
21
+ results.tsv 1345
README.md CHANGED
@@ -26,7 +26,7 @@ Until Laguna support lands in the runners you use, these files should be treated
26
  | `IQ3_K_R4` | `ik-spark-core` | finished | 91.81 GiB | 2.4770 +/- 0.04102 | pending | Better quality while still on the ik CUDA fast path. |
27
  | `IQ4_KS_R4` | `ik-spark-core` | finished | 113.20 GiB | 2.4140 +/- 0.03950 | pending | Strong quality while still optimized for ik CUDA. |
28
  | `IQ3_S` | `vanilla-core` | finished | 91.81 GiB | 2.4683 +/- 0.04051 | pending | Vanilla CUDA MMQ-supported imatrix low-bit baseline. |
29
- | `Q4_K_M` | `vanilla-core` | pending | pending | pending | pending | Broadly compatible CUDA-supported 4-bit-ish baseline. |
30
  | `IQ4_XS` | `vanilla-core` | pending | pending | pending | pending | Vanilla CUDA-supported nonlinear 4-bit-ish candidate. |
31
  | `Q5_K_M` | `vanilla-core` | pending | pending | pending | pending | Higher-quality broadly compatible reference. |
32
  | `Q6_K` | `vanilla-optional` | pending | pending | pending | pending | Optional near-reference quant if pool/upload budget allows. |
 
26
  | `IQ3_K_R4` | `ik-spark-core` | finished | 91.81 GiB | 2.4770 +/- 0.04102 | pending | Better quality while still on the ik CUDA fast path. |
27
  | `IQ4_KS_R4` | `ik-spark-core` | finished | 113.20 GiB | 2.4140 +/- 0.03950 | pending | Strong quality while still optimized for ik CUDA. |
28
  | `IQ3_S` | `vanilla-core` | finished | 91.81 GiB | 2.4683 +/- 0.04051 | pending | Vanilla CUDA MMQ-supported imatrix low-bit baseline. |
29
+ | `Q4_K_M` | `vanilla-core` | finished | 127.33 GiB | 2.4084 +/- 0.03947 | pending | Broadly compatible CUDA-supported 4-bit-ish baseline. |
30
  | `IQ4_XS` | `vanilla-core` | pending | pending | pending | pending | Vanilla CUDA-supported nonlinear 4-bit-ish candidate. |
31
  | `Q5_K_M` | `vanilla-core` | pending | pending | pending | pending | Higher-quality broadly compatible reference. |
32
  | `Q6_K` | `vanilla-optional` | pending | pending | pending | pending | Optional near-reference quant if pool/upload budget allows. |
results.tsv CHANGED
@@ -2,4 +2,5 @@ quant lane status files bytes sha256sums ppl ppl_unc kld kld_unc kl_reference co
2
  IQ3_K_R4 ik-spark-core finished 10 98576647840 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_K_R4/SHA256SUMS 2.4770 0.04102 pending pending pending 2026-06-21T01:43:29-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_K_R4/quantize.log
3
  IQ3_S vanilla-core finished 10 98576647840 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/SHA256SUMS 2.4683 0.04051 pending pending pending 2026-06-21T02:40:21-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/quantize.log
4
  IQ4_KS_R4 ik-spark-core finished 10 121548351136 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ4_KS_R4/SHA256SUMS 2.4140 0.03950 pending pending pending 2026-06-21T01:14:37-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ4_KS_R4/quantize.log
 
5
  IQ2_K_R4 ik-spark-core finished 10 69096687264 /mnt/pool/gguf/laguna-m1/quants/public-v1/IQ2_K_R4/SHA256SUMS 2.6686 0.04561 pending pending pending 2026-06-20T22:58:57-04:00 /mnt/pool/gguf/laguna-m1/quants/public-v1/IQ2_K_R4/quantize.log
 
2
  IQ3_K_R4 ik-spark-core finished 10 98576647840 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_K_R4/SHA256SUMS 2.4770 0.04102 pending pending pending 2026-06-21T01:43:29-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_K_R4/quantize.log
3
  IQ3_S vanilla-core finished 10 98576647840 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/SHA256SUMS 2.4683 0.04051 pending pending pending 2026-06-21T02:40:21-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ3_S/quantize.log
4
  IQ4_KS_R4 ik-spark-core finished 10 121548351136 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ4_KS_R4/SHA256SUMS 2.4140 0.03950 pending pending pending 2026-06-21T01:14:37-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/IQ4_KS_R4/quantize.log
5
+ Q4_K_M vanilla-core finished 10 136723580576 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/Q4_K_M/SHA256SUMS 2.4084 0.03947 pending pending pending 2026-06-21T04:20:35-04:00 /home/janet/laguna-gguf/laguna-m1/quants/public-v1/Q4_K_M/quantize.log
6
  IQ2_K_R4 ik-spark-core finished 10 69096687264 /mnt/pool/gguf/laguna-m1/quants/public-v1/IQ2_K_R4/SHA256SUMS 2.6686 0.04561 pending pending pending 2026-06-20T22:58:57-04:00 /mnt/pool/gguf/laguna-m1/quants/public-v1/IQ2_K_R4/quantize.log