{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 3095, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.03231017770597738, "grad_norm": 2.219433546066284, "learning_rate": 3.166397415185784e-06, "loss": 0.13468141555786134, "step": 100 }, { "epoch": 0.06462035541195477, "grad_norm": 1.7088854312896729, "learning_rate": 6.397415185783522e-06, "loss": 0.08345352172851563, "step": 200 }, { "epoch": 0.09693053311793215, "grad_norm": 1.8784979581832886, "learning_rate": 9.62843295638126e-06, "loss": 0.07643356323242187, "step": 300 }, { "epoch": 0.12924071082390953, "grad_norm": 3.160496234893799, "learning_rate": 1.2859450726979e-05, "loss": 0.07728703022003174, "step": 400 }, { "epoch": 0.16155088852988692, "grad_norm": 1.793244481086731, "learning_rate": 1.609046849757674e-05, "loss": 0.0715451717376709, "step": 500 }, { "epoch": 0.1938610662358643, "grad_norm": 0.49027326703071594, "learning_rate": 1.9321486268174476e-05, "loss": 0.07877041339874268, "step": 600 }, { "epoch": 0.22617124394184168, "grad_norm": 0.728702962398529, "learning_rate": 1.999007830695722e-05, "loss": 0.07102310180664062, "step": 700 }, { "epoch": 0.25848142164781907, "grad_norm": 2.55074405670166, "learning_rate": 1.9949097313414066e-05, "loss": 0.074349045753479, "step": 800 }, { "epoch": 0.29079159935379645, "grad_norm": 3.9224605560302734, "learning_rate": 1.9876486114300215e-05, "loss": 0.07766849994659424, "step": 900 }, { "epoch": 0.32310177705977383, "grad_norm": 0.9257252216339111, "learning_rate": 1.9772475555398188e-05, "loss": 0.0694255781173706, "step": 1000 }, { "epoch": 0.3554119547657512, "grad_norm": 0.8081497550010681, "learning_rate": 1.9637396307446846e-05, "loss": 0.07494896411895752, "step": 1100 }, { "epoch": 0.3877221324717286, "grad_norm": 2.14691424369812, "learning_rate": 1.9471677814871786e-05, "loss": 0.0675746250152588, "step": 1200 }, { "epoch": 0.420032310177706, "grad_norm": 0.38400450348854065, "learning_rate": 1.927584693049412e-05, "loss": 0.07241044998168945, "step": 1300 }, { "epoch": 0.45234248788368336, "grad_norm": 0.580016553401947, "learning_rate": 1.9050526240558083e-05, "loss": 0.07040606498718262, "step": 1400 }, { "epoch": 0.48465266558966075, "grad_norm": 1.0501762628555298, "learning_rate": 1.8796432085402662e-05, "loss": 0.07063678741455078, "step": 1500 }, { "epoch": 0.5169628432956381, "grad_norm": 0.9808917045593262, "learning_rate": 1.8514372282069805e-05, "loss": 0.067080078125, "step": 1600 }, { "epoch": 0.5492730210016155, "grad_norm": 0.8777210116386414, "learning_rate": 1.8205243556089643e-05, "loss": 0.06770069122314454, "step": 1700 }, { "epoch": 0.5815831987075929, "grad_norm": 1.6874562501907349, "learning_rate": 1.7870028690607476e-05, "loss": 0.07079707622528077, "step": 1800 }, { "epoch": 0.6138933764135702, "grad_norm": 1.3245418071746826, "learning_rate": 1.7509793401916104e-05, "loss": 0.06404186248779296, "step": 1900 }, { "epoch": 0.6462035541195477, "grad_norm": 0.3585558831691742, "learning_rate": 1.7125682951326795e-05, "loss": 0.06824737071990966, "step": 2000 }, { "epoch": 0.678513731825525, "grad_norm": 0.8804681897163391, "learning_rate": 1.671891850415046e-05, "loss": 0.07004157543182372, "step": 2100 }, { "epoch": 0.7108239095315024, "grad_norm": 0.6549601554870605, "learning_rate": 1.629079324736454e-05, "loss": 0.06939639568328858, "step": 2200 }, { "epoch": 0.7431340872374798, "grad_norm": 1.4602007865905762, "learning_rate": 1.584266827830838e-05, "loss": 0.06469368457794189, "step": 2300 }, { "epoch": 0.7754442649434572, "grad_norm": 1.0985392332077026, "learning_rate": 1.537596827747772e-05, "loss": 0.07550141334533692, "step": 2400 }, { "epoch": 0.8077544426494345, "grad_norm": 0.9523833394050598, "learning_rate": 1.4892176979175388e-05, "loss": 0.06276164531707763, "step": 2500 }, { "epoch": 0.840064620355412, "grad_norm": 0.4565122723579407, "learning_rate": 1.4392832454417938e-05, "loss": 0.0644590950012207, "step": 2600 }, { "epoch": 0.8723747980613893, "grad_norm": 2.104994058609009, "learning_rate": 1.387952222109479e-05, "loss": 0.06694219589233398, "step": 2700 }, { "epoch": 0.9046849757673667, "grad_norm": 1.4507120847702026, "learning_rate": 1.3353878196925727e-05, "loss": 0.059137086868286136, "step": 2800 }, { "epoch": 0.9369951534733441, "grad_norm": 1.1051520109176636, "learning_rate": 1.2817571511262256e-05, "loss": 0.06713937282562256, "step": 2900 }, { "epoch": 0.9693053311793215, "grad_norm": 1.7327321767807007, "learning_rate": 1.2272307192227245e-05, "loss": 0.0657884407043457, "step": 3000 } ], "logging_steps": 100, "max_steps": 6190, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 16, "trial_name": null, "trial_params": null }