{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.01171929361957708, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0002343858723915416, "grad_norm": 7.478665351867676, "learning_rate": 1.40625e-07, "loss": 0.6732528686523438, "step": 10 }, { "epoch": 0.0004687717447830832, "grad_norm": 7.584623336791992, "learning_rate": 2.96875e-07, "loss": 0.66597900390625, "step": 20 }, { "epoch": 0.0007031576171746248, "grad_norm": 6.306713581085205, "learning_rate": 4.53125e-07, "loss": 0.6197677612304687, "step": 30 }, { "epoch": 0.0009375434895661664, "grad_norm": 4.607383728027344, "learning_rate": 6.09375e-07, "loss": 0.5319419860839844, "step": 40 }, { "epoch": 0.001171929361957708, "grad_norm": 2.3062853813171387, "learning_rate": 7.656250000000001e-07, "loss": 0.42061538696289064, "step": 50 }, { "epoch": 0.0014063152343492496, "grad_norm": 1.4863656759262085, "learning_rate": 9.218750000000002e-07, "loss": 0.3130531311035156, "step": 60 }, { "epoch": 0.0016407011067407912, "grad_norm": 0.8237831592559814, "learning_rate": 1.0781250000000002e-06, "loss": 0.23311138153076172, "step": 70 }, { "epoch": 0.0018750869791323327, "grad_norm": 0.6643654704093933, "learning_rate": 1.2343750000000001e-06, "loss": 0.2020857810974121, "step": 80 }, { "epoch": 0.0021094728515238742, "grad_norm": 0.5149619579315186, "learning_rate": 1.3906250000000001e-06, "loss": 0.18220624923706055, "step": 90 }, { "epoch": 0.002343858723915416, "grad_norm": 0.5113442540168762, "learning_rate": 1.5468750000000001e-06, "loss": 0.16905088424682618, "step": 100 }, { "epoch": 0.0025782445963069577, "grad_norm": 0.5090538859367371, "learning_rate": 1.703125e-06, "loss": 0.16316838264465333, "step": 110 }, { "epoch": 0.0028126304686984993, "grad_norm": 0.4110700190067291, "learning_rate": 1.8593750000000003e-06, "loss": 0.1580258846282959, "step": 120 }, { "epoch": 0.003047016341090041, "grad_norm": 0.5005377531051636, "learning_rate": 2.0156250000000003e-06, "loss": 0.14841842651367188, "step": 130 }, { "epoch": 0.0032814022134815823, "grad_norm": 0.4914725124835968, "learning_rate": 2.1718750000000003e-06, "loss": 0.14541361331939698, "step": 140 }, { "epoch": 0.003515788085873124, "grad_norm": 0.5155150294303894, "learning_rate": 2.3281250000000003e-06, "loss": 0.14492216110229492, "step": 150 }, { "epoch": 0.0037501739582646654, "grad_norm": 0.4690794050693512, "learning_rate": 2.4843750000000002e-06, "loss": 0.14195268154144286, "step": 160 }, { "epoch": 0.003984559830656207, "grad_norm": 0.50107342004776, "learning_rate": 2.640625e-06, "loss": 0.1428708553314209, "step": 170 }, { "epoch": 0.0042189457030477485, "grad_norm": 0.48859548568725586, "learning_rate": 2.796875e-06, "loss": 0.14226298332214354, "step": 180 }, { "epoch": 0.0044533315754392904, "grad_norm": 0.5123774409294128, "learning_rate": 2.953125e-06, "loss": 0.1370567798614502, "step": 190 }, { "epoch": 0.004687717447830832, "grad_norm": 0.4833315312862396, "learning_rate": 3.109375e-06, "loss": 0.1390127420425415, "step": 200 }, { "epoch": 0.0049221033202223735, "grad_norm": 0.43915465474128723, "learning_rate": 3.265625e-06, "loss": 0.13475892543792725, "step": 210 }, { "epoch": 0.0051564891926139155, "grad_norm": 0.46104806661605835, "learning_rate": 3.421875e-06, "loss": 0.13139626979827881, "step": 220 }, { "epoch": 0.005390875065005457, "grad_norm": 0.5122255086898804, "learning_rate": 3.578125e-06, "loss": 0.13110661506652832, "step": 230 }, { "epoch": 0.0056252609373969985, "grad_norm": 0.48517927527427673, "learning_rate": 3.734375e-06, "loss": 0.12811717987060547, "step": 240 }, { "epoch": 0.00585964680978854, "grad_norm": 0.4835013449192047, "learning_rate": 3.890625e-06, "loss": 0.13097090721130372, "step": 250 }, { "epoch": 0.006094032682180082, "grad_norm": 0.4820455312728882, "learning_rate": 4.046875e-06, "loss": 0.12818758487701415, "step": 260 }, { "epoch": 0.006328418554571624, "grad_norm": 0.44813403487205505, "learning_rate": 4.2031250000000005e-06, "loss": 0.12801458835601806, "step": 270 }, { "epoch": 0.006562804426963165, "grad_norm": 0.5279086232185364, "learning_rate": 4.359375e-06, "loss": 0.12751049995422364, "step": 280 }, { "epoch": 0.006797190299354707, "grad_norm": 0.42281895875930786, "learning_rate": 4.5156250000000005e-06, "loss": 0.13442916870117189, "step": 290 }, { "epoch": 0.007031576171746248, "grad_norm": 0.4167884588241577, "learning_rate": 4.671875e-06, "loss": 0.12415478229522706, "step": 300 }, { "epoch": 0.00726596204413779, "grad_norm": 0.4328134059906006, "learning_rate": 4.8281250000000005e-06, "loss": 0.12891750335693358, "step": 310 }, { "epoch": 0.007500347916529331, "grad_norm": 0.45806413888931274, "learning_rate": 4.984375e-06, "loss": 0.12102985382080078, "step": 320 }, { "epoch": 0.007734733788920873, "grad_norm": 0.5078896880149841, "learning_rate": 5.1406250000000004e-06, "loss": 0.12243216037750244, "step": 330 }, { "epoch": 0.007969119661312415, "grad_norm": 0.527930498123169, "learning_rate": 5.296875e-06, "loss": 0.12513620853424073, "step": 340 }, { "epoch": 0.008203505533703956, "grad_norm": 0.431318461894989, "learning_rate": 5.453125e-06, "loss": 0.12151198387145996, "step": 350 }, { "epoch": 0.008437891406095497, "grad_norm": 0.4996262788772583, "learning_rate": 5.609375e-06, "loss": 0.11847388744354248, "step": 360 }, { "epoch": 0.00867227727848704, "grad_norm": 0.4818030893802643, "learning_rate": 5.765625e-06, "loss": 0.1227838158607483, "step": 370 }, { "epoch": 0.008906663150878581, "grad_norm": 0.4577305018901825, "learning_rate": 5.921875e-06, "loss": 0.1199771523475647, "step": 380 }, { "epoch": 0.009141049023270122, "grad_norm": 0.48360294103622437, "learning_rate": 6.078125e-06, "loss": 0.11903550624847412, "step": 390 }, { "epoch": 0.009375434895661665, "grad_norm": 0.47577178478240967, "learning_rate": 6.234375e-06, "loss": 0.12074012756347656, "step": 400 }, { "epoch": 0.009609820768053206, "grad_norm": 0.4833427369594574, "learning_rate": 6.390625e-06, "loss": 0.12326394319534302, "step": 410 }, { "epoch": 0.009844206640444747, "grad_norm": 0.43909400701522827, "learning_rate": 6.546875e-06, "loss": 0.11921967267990112, "step": 420 }, { "epoch": 0.010078592512836288, "grad_norm": 0.5028750896453857, "learning_rate": 6.703125e-06, "loss": 0.12188574075698852, "step": 430 }, { "epoch": 0.010312978385227831, "grad_norm": 0.47389668226242065, "learning_rate": 6.859375000000001e-06, "loss": 0.1192385196685791, "step": 440 }, { "epoch": 0.010547364257619372, "grad_norm": 0.5239953398704529, "learning_rate": 7.015625e-06, "loss": 0.11499193906784058, "step": 450 }, { "epoch": 0.010781750130010913, "grad_norm": 0.5061202645301819, "learning_rate": 7.171875000000001e-06, "loss": 0.11584588289260864, "step": 460 }, { "epoch": 0.011016136002402456, "grad_norm": 0.4635751247406006, "learning_rate": 7.328125e-06, "loss": 0.11434779167175294, "step": 470 }, { "epoch": 0.011250521874793997, "grad_norm": 0.4353443682193756, "learning_rate": 7.484375000000001e-06, "loss": 0.11755821704864503, "step": 480 }, { "epoch": 0.011484907747185538, "grad_norm": 0.41028693318367004, "learning_rate": 7.640625000000001e-06, "loss": 0.11704204082489014, "step": 490 }, { "epoch": 0.01171929361957708, "grad_norm": 0.4426499009132385, "learning_rate": 7.796875e-06, "loss": 0.11789888143539429, "step": 500 } ], "logging_steps": 10, "max_steps": 42665, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 2.1936814887618478e+18, "train_batch_size": 1, "trial_name": null, "trial_params": null }