autotrain-roblox-10 / checkpoint-300 /trainer_state.json
trentmkelly's picture
Upload folder using huggingface_hub
2743640 verified
Raw
History Blame Contribute Delete
9.31 kB
{
"best_metric": 0.40960419178009033,
"best_model_checkpoint": "autotrain-roblox-10/checkpoint-300",
"epoch": 1.0,
"eval_steps": 500,
"global_step": 300,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.02,
"grad_norm": 8.799605369567871,
"learning_rate": 2.777777777777778e-06,
"loss": 0.8052,
"step": 6
},
{
"epoch": 0.04,
"grad_norm": 11.565935134887695,
"learning_rate": 6.111111111111111e-06,
"loss": 0.8027,
"step": 12
},
{
"epoch": 0.06,
"grad_norm": 7.1787590980529785,
"learning_rate": 9.444444444444445e-06,
"loss": 0.796,
"step": 18
},
{
"epoch": 0.08,
"grad_norm": 2.775609254837036,
"learning_rate": 1.2777777777777777e-05,
"loss": 0.7706,
"step": 24
},
{
"epoch": 0.1,
"grad_norm": 4.121618747711182,
"learning_rate": 1.6111111111111115e-05,
"loss": 0.7175,
"step": 30
},
{
"epoch": 0.12,
"grad_norm": 1.8891937732696533,
"learning_rate": 1.9444444444444445e-05,
"loss": 0.697,
"step": 36
},
{
"epoch": 0.14,
"grad_norm": 2.095088243484497,
"learning_rate": 2.277777777777778e-05,
"loss": 0.6848,
"step": 42
},
{
"epoch": 0.16,
"grad_norm": 3.5387184619903564,
"learning_rate": 2.6111111111111114e-05,
"loss": 0.7146,
"step": 48
},
{
"epoch": 0.18,
"grad_norm": 4.3815131187438965,
"learning_rate": 2.9444444444444448e-05,
"loss": 0.6832,
"step": 54
},
{
"epoch": 0.2,
"grad_norm": 6.097675800323486,
"learning_rate": 3.277777777777778e-05,
"loss": 0.7085,
"step": 60
},
{
"epoch": 0.22,
"grad_norm": 3.172816276550293,
"learning_rate": 3.611111111111111e-05,
"loss": 0.6388,
"step": 66
},
{
"epoch": 0.24,
"grad_norm": 2.8284499645233154,
"learning_rate": 3.944444444444445e-05,
"loss": 0.727,
"step": 72
},
{
"epoch": 0.26,
"grad_norm": 4.939257621765137,
"learning_rate": 4.277777777777778e-05,
"loss": 0.653,
"step": 78
},
{
"epoch": 0.28,
"grad_norm": 5.6476521492004395,
"learning_rate": 4.6111111111111115e-05,
"loss": 0.6342,
"step": 84
},
{
"epoch": 0.3,
"grad_norm": 8.957379341125488,
"learning_rate": 4.9444444444444446e-05,
"loss": 0.5829,
"step": 90
},
{
"epoch": 0.32,
"grad_norm": 10.431815147399902,
"learning_rate": 4.969135802469136e-05,
"loss": 0.629,
"step": 96
},
{
"epoch": 0.34,
"grad_norm": 7.197211742401123,
"learning_rate": 4.938271604938271e-05,
"loss": 0.6038,
"step": 102
},
{
"epoch": 0.36,
"grad_norm": 18.572933197021484,
"learning_rate": 4.901234567901235e-05,
"loss": 0.5653,
"step": 108
},
{
"epoch": 0.38,
"grad_norm": 3.8738515377044678,
"learning_rate": 4.864197530864198e-05,
"loss": 0.5197,
"step": 114
},
{
"epoch": 0.4,
"grad_norm": 13.589826583862305,
"learning_rate": 4.827160493827161e-05,
"loss": 0.5653,
"step": 120
},
{
"epoch": 0.42,
"grad_norm": 7.053980350494385,
"learning_rate": 4.7901234567901237e-05,
"loss": 0.6216,
"step": 126
},
{
"epoch": 0.44,
"grad_norm": 5.756833076477051,
"learning_rate": 4.7530864197530866e-05,
"loss": 0.6131,
"step": 132
},
{
"epoch": 0.46,
"grad_norm": 4.334901809692383,
"learning_rate": 4.7160493827160495e-05,
"loss": 0.5738,
"step": 138
},
{
"epoch": 0.48,
"grad_norm": 5.856199264526367,
"learning_rate": 4.6790123456790124e-05,
"loss": 0.5103,
"step": 144
},
{
"epoch": 0.5,
"grad_norm": 5.805249214172363,
"learning_rate": 4.641975308641975e-05,
"loss": 0.4794,
"step": 150
},
{
"epoch": 0.52,
"grad_norm": 6.147461414337158,
"learning_rate": 4.604938271604938e-05,
"loss": 0.4606,
"step": 156
},
{
"epoch": 0.54,
"grad_norm": 10.369290351867676,
"learning_rate": 4.567901234567901e-05,
"loss": 0.4953,
"step": 162
},
{
"epoch": 0.56,
"grad_norm": 12.811848640441895,
"learning_rate": 4.530864197530865e-05,
"loss": 0.3625,
"step": 168
},
{
"epoch": 0.58,
"grad_norm": 11.082608222961426,
"learning_rate": 4.493827160493828e-05,
"loss": 0.5738,
"step": 174
},
{
"epoch": 0.6,
"grad_norm": 2.9676032066345215,
"learning_rate": 4.4567901234567906e-05,
"loss": 0.374,
"step": 180
},
{
"epoch": 0.62,
"grad_norm": 19.597843170166016,
"learning_rate": 4.4197530864197535e-05,
"loss": 0.4335,
"step": 186
},
{
"epoch": 0.64,
"grad_norm": 6.006842613220215,
"learning_rate": 4.3827160493827164e-05,
"loss": 0.5708,
"step": 192
},
{
"epoch": 0.66,
"grad_norm": 6.002368450164795,
"learning_rate": 4.345679012345679e-05,
"loss": 0.4244,
"step": 198
},
{
"epoch": 0.68,
"grad_norm": 4.788684368133545,
"learning_rate": 4.308641975308642e-05,
"loss": 0.4441,
"step": 204
},
{
"epoch": 0.7,
"grad_norm": 6.1746954917907715,
"learning_rate": 4.271604938271605e-05,
"loss": 0.5917,
"step": 210
},
{
"epoch": 0.72,
"grad_norm": 4.382603645324707,
"learning_rate": 4.234567901234568e-05,
"loss": 0.5596,
"step": 216
},
{
"epoch": 0.74,
"grad_norm": 30.41250228881836,
"learning_rate": 4.197530864197531e-05,
"loss": 0.521,
"step": 222
},
{
"epoch": 0.76,
"grad_norm": 4.90874719619751,
"learning_rate": 4.1604938271604946e-05,
"loss": 0.4642,
"step": 228
},
{
"epoch": 0.78,
"grad_norm": 7.5287675857543945,
"learning_rate": 4.1234567901234575e-05,
"loss": 0.4263,
"step": 234
},
{
"epoch": 0.8,
"grad_norm": 10.244704246520996,
"learning_rate": 4.0864197530864204e-05,
"loss": 0.5723,
"step": 240
},
{
"epoch": 0.82,
"grad_norm": 4.225964069366455,
"learning_rate": 4.049382716049383e-05,
"loss": 0.4529,
"step": 246
},
{
"epoch": 0.84,
"grad_norm": 4.297839641571045,
"learning_rate": 4.012345679012346e-05,
"loss": 0.4485,
"step": 252
},
{
"epoch": 0.86,
"grad_norm": 8.949761390686035,
"learning_rate": 3.975308641975309e-05,
"loss": 0.463,
"step": 258
},
{
"epoch": 0.88,
"grad_norm": 5.039264678955078,
"learning_rate": 3.938271604938272e-05,
"loss": 0.446,
"step": 264
},
{
"epoch": 0.9,
"grad_norm": 5.201369762420654,
"learning_rate": 3.901234567901234e-05,
"loss": 0.4966,
"step": 270
},
{
"epoch": 0.92,
"grad_norm": 7.023820400238037,
"learning_rate": 3.864197530864197e-05,
"loss": 0.5405,
"step": 276
},
{
"epoch": 0.94,
"grad_norm": 5.257862091064453,
"learning_rate": 3.82716049382716e-05,
"loss": 0.4393,
"step": 282
},
{
"epoch": 0.96,
"grad_norm": 9.458020210266113,
"learning_rate": 3.790123456790123e-05,
"loss": 0.3889,
"step": 288
},
{
"epoch": 0.98,
"grad_norm": 6.277890205383301,
"learning_rate": 3.7530864197530867e-05,
"loss": 0.4674,
"step": 294
},
{
"epoch": 1.0,
"grad_norm": 5.080256462097168,
"learning_rate": 3.7160493827160496e-05,
"loss": 0.4566,
"step": 300
},
{
"epoch": 1.0,
"eval_accuracy": 0.8067542213883677,
"eval_auc": 0.8871317829457365,
"eval_f1": 0.8074766355140187,
"eval_loss": 0.40960419178009033,
"eval_precision": 0.779783393501805,
"eval_recall": 0.8372093023255814,
"eval_runtime": 0.9868,
"eval_samples_per_second": 540.12,
"eval_steps_per_second": 17.227,
"step": 300
}
],
"logging_steps": 6,
"max_steps": 900,
"num_input_tokens_seen": 0,
"num_train_epochs": 3,
"save_steps": 500,
"stateful_callbacks": {
"EarlyStoppingCallback": {
"args": {
"early_stopping_patience": 5,
"early_stopping_threshold": 0.01
},
"attributes": {
"early_stopping_patience_counter": 0
}
},
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 315272822085120.0,
"train_batch_size": 16,
"trial_name": null,
"trial_params": null
}