kallacharanteja commited on
Commit
afab9b6
·
verified ·
1 Parent(s): 7f02a14

Training in progress, step 5664

Browse files
Files changed (2) hide show
  1. adapter_model.safetensors +1 -1
  2. trainer_state.json +35 -5
adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:53d3f4f4fca5e6d361920216050960841df09504c1be33e101bcf16c4080179c
3
  size 3089143504
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5aacb3f9f436b2e23a3db07bc2dd8c9f311053d3533138f47470082d0efd3eff
3
  size 3089143504
trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": 5500,
3
  "best_metric": 1.8417972326278687,
4
  "best_model_checkpoint": "/kaggle/working/checkpoints/checkpoint-5500",
5
- "epoch": 5.82692817386695,
6
  "eval_steps": 500,
7
- "global_step": 5500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1239,11 +1239,41 @@
1239
  "eval_steps_per_second": 1.768,
1240
  "num_input_tokens_seen": 22510080,
1241
  "step": 5500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1242
  }
1243
  ],
1244
  "logging_steps": 50,
1245
  "max_steps": 5664,
1246
- "num_input_tokens_seen": 22510080,
1247
  "num_train_epochs": 6,
1248
  "save_steps": 500,
1249
  "stateful_callbacks": {
@@ -1253,12 +1283,12 @@
1253
  "should_evaluate": false,
1254
  "should_log": false,
1255
  "should_save": true,
1256
- "should_training_stop": false
1257
  },
1258
  "attributes": {}
1259
  }
1260
  },
1261
- "total_flos": 4.107608856723456e+16,
1262
  "train_batch_size": 8,
1263
  "trial_name": null,
1264
  "trial_params": null
 
2
  "best_global_step": 5500,
3
  "best_metric": 1.8417972326278687,
4
  "best_model_checkpoint": "/kaggle/working/checkpoints/checkpoint-5500",
5
+ "epoch": 6.0,
6
  "eval_steps": 500,
7
+ "global_step": 5664,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1239
  "eval_steps_per_second": 1.768,
1240
  "num_input_tokens_seen": 22510080,
1241
  "step": 5500
1242
+ },
1243
+ {
1244
+ "epoch": 5.879936390140472,
1245
+ "grad_norm": 0.7171221375465393,
1246
+ "learning_rate": 6.279577721150346e-06,
1247
+ "loss": 8.624059448242187,
1248
+ "num_input_tokens_seen": 22714880,
1249
+ "step": 5550,
1250
+ "train_runtime": 11215.8439,
1251
+ "train_tokens_per_second": 2025.249
1252
+ },
1253
+ {
1254
+ "epoch": 5.932944606413994,
1255
+ "grad_norm": 0.6685547232627869,
1256
+ "learning_rate": 3.549326538041499e-06,
1257
+ "loss": 8.639024047851562,
1258
+ "num_input_tokens_seen": 22919680,
1259
+ "step": 5600,
1260
+ "train_runtime": 11307.5506,
1261
+ "train_tokens_per_second": 2026.936
1262
+ },
1263
+ {
1264
+ "epoch": 5.9859528226875165,
1265
+ "grad_norm": 0.6983809471130371,
1266
+ "learning_rate": 8.190753549326538e-07,
1267
+ "loss": 8.527723388671875,
1268
+ "num_input_tokens_seen": 23124480,
1269
+ "step": 5650,
1270
+ "train_runtime": 11399.0902,
1271
+ "train_tokens_per_second": 2028.625
1272
  }
1273
  ],
1274
  "logging_steps": 50,
1275
  "max_steps": 5664,
1276
+ "num_input_tokens_seen": 23178240,
1277
  "num_train_epochs": 6,
1278
  "save_steps": 500,
1279
  "stateful_callbacks": {
 
1283
  "should_evaluate": false,
1284
  "should_log": false,
1285
  "should_save": true,
1286
+ "should_training_stop": true
1287
  },
1288
  "attributes": {}
1289
  }
1290
  },
1291
+ "total_flos": 4.229533786963968e+16,
1292
  "train_batch_size": 8,
1293
  "trial_name": null,
1294
  "trial_params": null