{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.01171929361957708, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0002343858723915416, "grad_norm": Infinity, "learning_rate": 1.40625e-07, "loss": 0.6092071533203125, "step": 10 }, { "epoch": 0.0004687717447830832, "grad_norm": Infinity, "learning_rate": 2.96875e-07, "loss": 0.6080734252929687, "step": 20 }, { "epoch": 0.0007031576171746248, "grad_norm": Infinity, "learning_rate": 4.53125e-07, "loss": 0.6008880615234375, "step": 30 }, { "epoch": 0.0009375434895661664, "grad_norm": Infinity, "learning_rate": 6.09375e-07, "loss": 0.6059707641601563, "step": 40 }, { "epoch": 0.001171929361957708, "grad_norm": Infinity, "learning_rate": 7.656250000000001e-07, "loss": 0.615313720703125, "step": 50 }, { "epoch": 0.0014063152343492496, "grad_norm": Infinity, "learning_rate": 9.218750000000002e-07, "loss": 0.6074554443359375, "step": 60 }, { "epoch": 0.0016407011067407912, "grad_norm": Infinity, "learning_rate": 1.0781250000000002e-06, "loss": 0.605792236328125, "step": 70 }, { "epoch": 0.0018750869791323327, "grad_norm": Infinity, "learning_rate": 1.2343750000000001e-06, "loss": 0.6068206787109375, "step": 80 }, { "epoch": 0.0021094728515238742, "grad_norm": Infinity, "learning_rate": 1.3906250000000001e-06, "loss": 0.6020843505859375, "step": 90 }, { "epoch": 0.002343858723915416, "grad_norm": Infinity, "learning_rate": 1.5468750000000001e-06, "loss": 0.59473876953125, "step": 100 }, { "epoch": 0.0025782445963069577, "grad_norm": Infinity, "learning_rate": 1.703125e-06, "loss": 0.6020477294921875, "step": 110 }, { "epoch": 0.0028126304686984993, "grad_norm": Infinity, "learning_rate": 1.8593750000000003e-06, "loss": 0.596124267578125, "step": 120 }, { "epoch": 0.003047016341090041, "grad_norm": Infinity, "learning_rate": 2.0156250000000003e-06, "loss": 0.6029708862304688, "step": 130 }, { "epoch": 0.0032814022134815823, "grad_norm": Infinity, "learning_rate": 2.1718750000000003e-06, "loss": 0.5990798950195313, "step": 140 }, { "epoch": 0.003515788085873124, "grad_norm": Infinity, "learning_rate": 2.3281250000000003e-06, "loss": 0.5956100463867188, "step": 150 }, { "epoch": 0.0037501739582646654, "grad_norm": Infinity, "learning_rate": 2.4843750000000002e-06, "loss": 0.59793701171875, "step": 160 }, { "epoch": 0.003984559830656207, "grad_norm": Infinity, "learning_rate": 2.640625e-06, "loss": 0.6025466918945312, "step": 170 }, { "epoch": 0.0042189457030477485, "grad_norm": Infinity, "learning_rate": 2.796875e-06, "loss": 0.6034042358398437, "step": 180 }, { "epoch": 0.0044533315754392904, "grad_norm": Infinity, "learning_rate": 2.953125e-06, "loss": 0.6109519958496094, "step": 190 }, { "epoch": 0.004687717447830832, "grad_norm": Infinity, "learning_rate": 3.109375e-06, "loss": 0.6124420166015625, "step": 200 }, { "epoch": 0.0049221033202223735, "grad_norm": Infinity, "learning_rate": 3.265625e-06, "loss": 0.5962295532226562, "step": 210 }, { "epoch": 0.0051564891926139155, "grad_norm": Infinity, "learning_rate": 3.421875e-06, "loss": 0.5996864318847657, "step": 220 }, { "epoch": 0.005390875065005457, "grad_norm": Infinity, "learning_rate": 3.578125e-06, "loss": 0.6116897583007812, "step": 230 }, { "epoch": 0.0056252609373969985, "grad_norm": Infinity, "learning_rate": 3.734375e-06, "loss": 0.6195701599121094, "step": 240 }, { "epoch": 0.00585964680978854, "grad_norm": Infinity, "learning_rate": 3.890625e-06, "loss": 0.6111648559570313, "step": 250 }, { "epoch": 0.006094032682180082, "grad_norm": Infinity, "learning_rate": 4.046875e-06, "loss": 0.6055526733398438, "step": 260 }, { "epoch": 0.006328418554571624, "grad_norm": Infinity, "learning_rate": 4.2031250000000005e-06, "loss": 0.5984527587890625, "step": 270 }, { "epoch": 0.006562804426963165, "grad_norm": Infinity, "learning_rate": 4.359375e-06, "loss": 0.6135429382324219, "step": 280 }, { "epoch": 0.006797190299354707, "grad_norm": Infinity, "learning_rate": 4.5156250000000005e-06, "loss": 0.5952308654785157, "step": 290 }, { "epoch": 0.007031576171746248, "grad_norm": Infinity, "learning_rate": 4.671875e-06, "loss": 0.6037254333496094, "step": 300 }, { "epoch": 0.00726596204413779, "grad_norm": Infinity, "learning_rate": 4.8281250000000005e-06, "loss": 0.5975120544433594, "step": 310 }, { "epoch": 0.007500347916529331, "grad_norm": Infinity, "learning_rate": 4.984375e-06, "loss": 0.5918365478515625, "step": 320 }, { "epoch": 0.007734733788920873, "grad_norm": Infinity, "learning_rate": 5.1406250000000004e-06, "loss": 0.6080474853515625, "step": 330 }, { "epoch": 0.007969119661312415, "grad_norm": Infinity, "learning_rate": 5.296875e-06, "loss": 0.6048797607421875, "step": 340 }, { "epoch": 0.008203505533703956, "grad_norm": Infinity, "learning_rate": 5.453125e-06, "loss": 0.593927001953125, "step": 350 }, { "epoch": 0.008437891406095497, "grad_norm": Infinity, "learning_rate": 5.609375e-06, "loss": 0.6036277770996094, "step": 360 }, { "epoch": 0.00867227727848704, "grad_norm": Infinity, "learning_rate": 5.765625e-06, "loss": 0.604638671875, "step": 370 }, { "epoch": 0.008906663150878581, "grad_norm": Infinity, "learning_rate": 5.921875e-06, "loss": 0.605169677734375, "step": 380 }, { "epoch": 0.009141049023270122, "grad_norm": Infinity, "learning_rate": 6.078125e-06, "loss": 0.613519287109375, "step": 390 }, { "epoch": 0.009375434895661665, "grad_norm": Infinity, "learning_rate": 6.234375e-06, "loss": 0.6017379760742188, "step": 400 }, { "epoch": 0.009609820768053206, "grad_norm": Infinity, "learning_rate": 6.390625e-06, "loss": 0.6115753173828125, "step": 410 }, { "epoch": 0.009844206640444747, "grad_norm": Infinity, "learning_rate": 6.546875e-06, "loss": 0.6092376708984375, "step": 420 }, { "epoch": 0.010078592512836288, "grad_norm": Infinity, "learning_rate": 6.703125e-06, "loss": 0.6099075317382813, "step": 430 }, { "epoch": 0.010312978385227831, "grad_norm": Infinity, "learning_rate": 6.859375000000001e-06, "loss": 0.61324462890625, "step": 440 }, { "epoch": 0.010547364257619372, "grad_norm": Infinity, "learning_rate": 7.015625e-06, "loss": 0.6022865295410156, "step": 450 }, { "epoch": 0.010781750130010913, "grad_norm": Infinity, "learning_rate": 7.171875000000001e-06, "loss": 0.6093353271484375, "step": 460 }, { "epoch": 0.011016136002402456, "grad_norm": Infinity, "learning_rate": 7.328125e-06, "loss": 0.62122802734375, "step": 470 }, { "epoch": 0.011250521874793997, "grad_norm": Infinity, "learning_rate": 7.484375000000001e-06, "loss": 0.6011276245117188, "step": 480 }, { "epoch": 0.011484907747185538, "grad_norm": Infinity, "learning_rate": 7.640625000000001e-06, "loss": 0.6096969604492187, "step": 490 }, { "epoch": 0.01171929361957708, "grad_norm": Infinity, "learning_rate": 7.796875e-06, "loss": 0.6085403442382813, "step": 500 } ], "logging_steps": 10, "max_steps": 42665, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 2.1936814887618478e+18, "train_batch_size": 1, "trial_name": null, "trial_params": null }