{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.6330365974282888, "eval_steps": 500.0, "global_step": 2000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0003165182987141444, "grad_norm": 424.0, "learning_rate": 3.79746835443038e-08, "loss": 12.429935455322266, "step": 1, "token_acc": 0.3053435114503817 }, { "epoch": 0.1582591493570722, "grad_norm": 0.84375, "learning_rate": 5.8099004948061835e-06, "loss": 1.251413960733968, "step": 500, "token_acc": 0.9084568173857424 }, { "epoch": 0.3165182987141444, "grad_norm": 0.5390625, "learning_rate": 4.908786401724081e-06, "loss": 0.010874405860900879, "step": 1000, "token_acc": 0.9976750193748385 }, { "epoch": 0.47477744807121663, "grad_norm": 0.09326171875, "learning_rate": 3.4968801808609607e-06, "loss": 0.009071314811706543, "step": 1500, "token_acc": 0.998050890042789 }, { "epoch": 0.6330365974282888, "grad_norm": 0.09326171875, "learning_rate": 1.952008592786012e-06, "loss": 0.006442168235778809, "step": 2000, "token_acc": 0.9988140489584917 } ], "logging_steps": 500, "max_steps": 3160, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 2000, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 4.332279735076086e+17, "train_batch_size": 2, "trial_name": null, "trial_params": null }