jeqcho commited on
Commit
446a2da
·
verified ·
1 Parent(s): 991e9df

Upload folder using huggingface_hub

Browse files
checkpoint-30/adapter_config.json CHANGED
@@ -34,13 +34,13 @@
34
  "rank_pattern": {},
35
  "revision": null,
36
  "target_modules": [
 
 
 
37
  "v_proj",
38
- "k_proj",
39
- "up_proj",
40
  "o_proj",
41
- "q_proj",
42
- "down_proj",
43
- "gate_proj"
44
  ],
45
  "target_parameters": null,
46
  "task_type": "CAUSAL_LM",
 
34
  "rank_pattern": {},
35
  "revision": null,
36
  "target_modules": [
37
+ "down_proj",
38
+ "q_proj",
39
+ "gate_proj",
40
  "v_proj",
 
 
41
  "o_proj",
42
+ "up_proj",
43
+ "k_proj"
 
44
  ],
45
  "target_parameters": null,
46
  "task_type": "CAUSAL_LM",
checkpoint-30/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:32cd276cc4085f634fdf4c79ed262b19f8f2a9b6eacb30a7c10341e5e71dca9e
3
  size 59933632
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fde086140e7963aedad838f8435896127787d98f02ecad970bc4ec15042a858a
3
  size 59933632
checkpoint-30/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b6ca7d76fb26b098a3e4f13e4a6690ebcfbfc36e63d805626b907788afb77000
3
  size 120166091
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7471b9291a078960c699328acf57efb81bc2eb2c5edecb4a448e58f710f9343f
3
  size 120166091
checkpoint-30/trainer_state.json CHANGED
@@ -67,156 +67,156 @@
67
  },
68
  {
69
  "epoch": 0.05934065934065934,
70
- "grad_norm": 0.8124693632125854,
71
  "learning_rate": 0.00019866962305986697,
72
  "loss": 1.3178,
73
  "step": 9
74
  },
75
  {
76
  "epoch": 0.06593406593406594,
77
- "grad_norm": 1.4746452569961548,
78
  "learning_rate": 0.00019822616407982261,
79
- "loss": 1.3747,
80
  "step": 10
81
  },
82
  {
83
  "epoch": 0.07252747252747253,
84
- "grad_norm": 1.0956214666366577,
85
  "learning_rate": 0.00019778270509977829,
86
- "loss": 1.2291,
87
  "step": 11
88
  },
89
  {
90
  "epoch": 0.07912087912087912,
91
- "grad_norm": 0.5584886074066162,
92
  "learning_rate": 0.00019733924611973393,
93
- "loss": 1.2929,
94
  "step": 12
95
  },
96
  {
97
  "epoch": 0.08571428571428572,
98
- "grad_norm": 0.8778825402259827,
99
  "learning_rate": 0.0001968957871396896,
100
- "loss": 1.0355,
101
  "step": 13
102
  },
103
  {
104
  "epoch": 0.09230769230769231,
105
- "grad_norm": 0.5020594596862793,
106
  "learning_rate": 0.00019645232815964525,
107
- "loss": 1.1668,
108
  "step": 14
109
  },
110
  {
111
  "epoch": 0.0989010989010989,
112
- "grad_norm": 0.4985601305961609,
113
  "learning_rate": 0.00019600886917960092,
114
- "loss": 1.1048,
115
  "step": 15
116
  },
117
  {
118
  "epoch": 0.1054945054945055,
119
- "grad_norm": 0.4597553312778473,
120
  "learning_rate": 0.00019556541019955653,
121
- "loss": 1.0144,
122
  "step": 16
123
  },
124
  {
125
  "epoch": 0.11208791208791209,
126
- "grad_norm": 0.44980549812316895,
127
  "learning_rate": 0.0001951219512195122,
128
- "loss": 1.128,
129
  "step": 17
130
  },
131
  {
132
  "epoch": 0.11868131868131868,
133
- "grad_norm": 0.33276212215423584,
134
  "learning_rate": 0.00019467849223946785,
135
- "loss": 0.9434,
136
  "step": 18
137
  },
138
  {
139
  "epoch": 0.12527472527472527,
140
- "grad_norm": 0.46772485971450806,
141
  "learning_rate": 0.00019423503325942352,
142
- "loss": 1.2302,
143
  "step": 19
144
  },
145
  {
146
  "epoch": 0.13186813186813187,
147
- "grad_norm": 0.3876091241836548,
148
  "learning_rate": 0.00019379157427937917,
149
- "loss": 1.0448,
150
  "step": 20
151
  },
152
  {
153
  "epoch": 0.13846153846153847,
154
- "grad_norm": 0.3786633610725403,
155
  "learning_rate": 0.00019334811529933484,
156
- "loss": 1.0449,
157
  "step": 21
158
  },
159
  {
160
  "epoch": 0.14505494505494507,
161
- "grad_norm": 0.4180683195590973,
162
  "learning_rate": 0.00019290465631929045,
163
- "loss": 1.1112,
164
  "step": 22
165
  },
166
  {
167
  "epoch": 0.15164835164835164,
168
- "grad_norm": 0.4051693081855774,
169
  "learning_rate": 0.00019246119733924613,
170
- "loss": 1.0586,
171
  "step": 23
172
  },
173
  {
174
  "epoch": 0.15824175824175823,
175
- "grad_norm": 0.30783703923225403,
176
  "learning_rate": 0.00019201773835920177,
177
- "loss": 0.9412,
178
  "step": 24
179
  },
180
  {
181
  "epoch": 0.16483516483516483,
182
- "grad_norm": 0.3420283794403076,
183
  "learning_rate": 0.00019157427937915744,
184
- "loss": 0.9213,
185
  "step": 25
186
  },
187
  {
188
  "epoch": 0.17142857142857143,
189
- "grad_norm": 0.47376200556755066,
190
  "learning_rate": 0.00019113082039911309,
191
- "loss": 0.9635,
192
  "step": 26
193
  },
194
  {
195
  "epoch": 0.17802197802197803,
196
- "grad_norm": 0.3626822233200073,
197
  "learning_rate": 0.00019068736141906876,
198
- "loss": 0.7696,
199
  "step": 27
200
  },
201
  {
202
  "epoch": 0.18461538461538463,
203
- "grad_norm": 0.46646377444267273,
204
  "learning_rate": 0.0001902439024390244,
205
- "loss": 1.098,
206
  "step": 28
207
  },
208
  {
209
  "epoch": 0.1912087912087912,
210
- "grad_norm": 0.38763922452926636,
211
  "learning_rate": 0.00018980044345898005,
212
- "loss": 1.1587,
213
  "step": 29
214
  },
215
  {
216
  "epoch": 0.1978021978021978,
217
- "grad_norm": 0.3439793288707733,
218
  "learning_rate": 0.00018935698447893572,
219
- "loss": 1.1153,
220
  "step": 30
221
  }
222
  ],
 
67
  },
68
  {
69
  "epoch": 0.05934065934065934,
70
+ "grad_norm": 0.812495768070221,
71
  "learning_rate": 0.00019866962305986697,
72
  "loss": 1.3178,
73
  "step": 9
74
  },
75
  {
76
  "epoch": 0.06593406593406594,
77
+ "grad_norm": 1.4647173881530762,
78
  "learning_rate": 0.00019822616407982261,
79
+ "loss": 1.3721,
80
  "step": 10
81
  },
82
  {
83
  "epoch": 0.07252747252747253,
84
+ "grad_norm": 1.0903385877609253,
85
  "learning_rate": 0.00019778270509977829,
86
+ "loss": 1.2288,
87
  "step": 11
88
  },
89
  {
90
  "epoch": 0.07912087912087912,
91
+ "grad_norm": 0.5566668510437012,
92
  "learning_rate": 0.00019733924611973393,
93
+ "loss": 1.2958,
94
  "step": 12
95
  },
96
  {
97
  "epoch": 0.08571428571428572,
98
+ "grad_norm": 0.9000307321548462,
99
  "learning_rate": 0.0001968957871396896,
100
+ "loss": 1.0353,
101
  "step": 13
102
  },
103
  {
104
  "epoch": 0.09230769230769231,
105
+ "grad_norm": 0.5002865791320801,
106
  "learning_rate": 0.00019645232815964525,
107
+ "loss": 1.1657,
108
  "step": 14
109
  },
110
  {
111
  "epoch": 0.0989010989010989,
112
+ "grad_norm": 0.4956091344356537,
113
  "learning_rate": 0.00019600886917960092,
114
+ "loss": 1.1026,
115
  "step": 15
116
  },
117
  {
118
  "epoch": 0.1054945054945055,
119
+ "grad_norm": 0.46295931935310364,
120
  "learning_rate": 0.00019556541019955653,
121
+ "loss": 1.0131,
122
  "step": 16
123
  },
124
  {
125
  "epoch": 0.11208791208791209,
126
+ "grad_norm": 0.44878455996513367,
127
  "learning_rate": 0.0001951219512195122,
128
+ "loss": 1.1302,
129
  "step": 17
130
  },
131
  {
132
  "epoch": 0.11868131868131868,
133
+ "grad_norm": 0.3325801193714142,
134
  "learning_rate": 0.00019467849223946785,
135
+ "loss": 0.9445,
136
  "step": 18
137
  },
138
  {
139
  "epoch": 0.12527472527472527,
140
+ "grad_norm": 0.46596264839172363,
141
  "learning_rate": 0.00019423503325942352,
142
+ "loss": 1.2325,
143
  "step": 19
144
  },
145
  {
146
  "epoch": 0.13186813186813187,
147
+ "grad_norm": 0.38645994663238525,
148
  "learning_rate": 0.00019379157427937917,
149
+ "loss": 1.0453,
150
  "step": 20
151
  },
152
  {
153
  "epoch": 0.13846153846153847,
154
+ "grad_norm": 0.3800373375415802,
155
  "learning_rate": 0.00019334811529933484,
156
+ "loss": 1.045,
157
  "step": 21
158
  },
159
  {
160
  "epoch": 0.14505494505494507,
161
+ "grad_norm": 0.4129738211631775,
162
  "learning_rate": 0.00019290465631929045,
163
+ "loss": 1.1104,
164
  "step": 22
165
  },
166
  {
167
  "epoch": 0.15164835164835164,
168
+ "grad_norm": 0.40392687916755676,
169
  "learning_rate": 0.00019246119733924613,
170
+ "loss": 1.0589,
171
  "step": 23
172
  },
173
  {
174
  "epoch": 0.15824175824175823,
175
+ "grad_norm": 0.3075926899909973,
176
  "learning_rate": 0.00019201773835920177,
177
+ "loss": 0.9411,
178
  "step": 24
179
  },
180
  {
181
  "epoch": 0.16483516483516483,
182
+ "grad_norm": 0.3475458323955536,
183
  "learning_rate": 0.00019157427937915744,
184
+ "loss": 0.9233,
185
  "step": 25
186
  },
187
  {
188
  "epoch": 0.17142857142857143,
189
+ "grad_norm": 0.49025434255599976,
190
  "learning_rate": 0.00019113082039911309,
191
+ "loss": 0.9665,
192
  "step": 26
193
  },
194
  {
195
  "epoch": 0.17802197802197803,
196
+ "grad_norm": 0.3652801215648651,
197
  "learning_rate": 0.00019068736141906876,
198
+ "loss": 0.7703,
199
  "step": 27
200
  },
201
  {
202
  "epoch": 0.18461538461538463,
203
+ "grad_norm": 0.45883381366729736,
204
  "learning_rate": 0.0001902439024390244,
205
+ "loss": 1.0975,
206
  "step": 28
207
  },
208
  {
209
  "epoch": 0.1912087912087912,
210
+ "grad_norm": 0.3874357044696808,
211
  "learning_rate": 0.00018980044345898005,
212
+ "loss": 1.1598,
213
  "step": 29
214
  },
215
  {
216
  "epoch": 0.1978021978021978,
217
+ "grad_norm": 0.34444889426231384,
218
  "learning_rate": 0.00018935698447893572,
219
+ "loss": 1.115,
220
  "step": 30
221
  }
222
  ],
checkpoint-30/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f1e6107b40730f00b0ecebeb53261d03c090314acc28d156f0c2a1f966f3cdca
3
  size 6417
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0f4aaa76910e776691a8e236eac6a84769f2e9172a54a9d30fe100706079516a
3
  size 6417