{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.12610340479192939, "eval_steps": 50, "global_step": 50, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0025220680958385876, "grad_norm": 83.50857543945312, "learning_rate": 0.0, "loss": 3.3887, "step": 1 }, { "epoch": 0.005044136191677175, "grad_norm": 71.31573486328125, "learning_rate": 2.0000000000000003e-06, "loss": 3.4305, "step": 2 }, { "epoch": 0.007566204287515763, "grad_norm": 66.55958557128906, "learning_rate": 4.000000000000001e-06, "loss": 3.0425, "step": 3 }, { "epoch": 0.01008827238335435, "grad_norm": 67.16094207763672, "learning_rate": 6e-06, "loss": 3.1193, "step": 4 }, { "epoch": 0.012610340479192938, "grad_norm": 58.159088134765625, "learning_rate": 8.000000000000001e-06, "loss": 2.7595, "step": 5 }, { "epoch": 0.015132408575031526, "grad_norm": 39.46601867675781, "learning_rate": 1e-05, "loss": 2.3184, "step": 6 }, { "epoch": 0.017654476670870115, "grad_norm": 45.5927734375, "learning_rate": 9.987325728770595e-06, "loss": 2.1763, "step": 7 }, { "epoch": 0.0201765447667087, "grad_norm": 20.798906326293945, "learning_rate": 9.974651457541193e-06, "loss": 2.1835, "step": 8 }, { "epoch": 0.02269861286254729, "grad_norm": 15.517325401306152, "learning_rate": 9.961977186311787e-06, "loss": 2.1354, "step": 9 }, { "epoch": 0.025220680958385876, "grad_norm": 15.667418479919434, "learning_rate": 9.949302915082384e-06, "loss": 1.9287, "step": 10 }, { "epoch": 0.027742749054224466, "grad_norm": 13.178864479064941, "learning_rate": 9.93662864385298e-06, "loss": 1.9898, "step": 11 }, { "epoch": 0.03026481715006305, "grad_norm": 10.053223609924316, "learning_rate": 9.923954372623576e-06, "loss": 2.0013, "step": 12 }, { "epoch": 0.03278688524590164, "grad_norm": 14.007344245910645, "learning_rate": 9.91128010139417e-06, "loss": 2.0559, "step": 13 }, { "epoch": 0.03530895334174023, "grad_norm": 11.14796257019043, "learning_rate": 9.898605830164766e-06, "loss": 2.0379, "step": 14 }, { "epoch": 0.03783102143757881, "grad_norm": 13.28112506866455, "learning_rate": 9.885931558935362e-06, "loss": 1.9854, "step": 15 }, { "epoch": 0.0403530895334174, "grad_norm": 11.45155143737793, "learning_rate": 9.873257287705957e-06, "loss": 2.1884, "step": 16 }, { "epoch": 0.04287515762925599, "grad_norm": 10.636513710021973, "learning_rate": 9.860583016476553e-06, "loss": 1.9676, "step": 17 }, { "epoch": 0.04539722572509458, "grad_norm": 10.700081825256348, "learning_rate": 9.847908745247149e-06, "loss": 1.9426, "step": 18 }, { "epoch": 0.04791929382093316, "grad_norm": 10.350312232971191, "learning_rate": 9.835234474017745e-06, "loss": 1.931, "step": 19 }, { "epoch": 0.05044136191677175, "grad_norm": 11.60483455657959, "learning_rate": 9.822560202788341e-06, "loss": 1.9397, "step": 20 }, { "epoch": 0.05296343001261034, "grad_norm": 10.923288345336914, "learning_rate": 9.809885931558936e-06, "loss": 1.8861, "step": 21 }, { "epoch": 0.05548549810844893, "grad_norm": 10.222322463989258, "learning_rate": 9.797211660329532e-06, "loss": 1.8154, "step": 22 }, { "epoch": 0.058007566204287514, "grad_norm": 9.665061950683594, "learning_rate": 9.784537389100128e-06, "loss": 1.9885, "step": 23 }, { "epoch": 0.0605296343001261, "grad_norm": 10.544888496398926, "learning_rate": 9.771863117870724e-06, "loss": 1.9728, "step": 24 }, { "epoch": 0.06305170239596469, "grad_norm": 9.924367904663086, "learning_rate": 9.759188846641318e-06, "loss": 1.9665, "step": 25 }, { "epoch": 0.06557377049180328, "grad_norm": 9.470415115356445, "learning_rate": 9.746514575411914e-06, "loss": 1.8872, "step": 26 }, { "epoch": 0.06809583858764187, "grad_norm": 9.330588340759277, "learning_rate": 9.73384030418251e-06, "loss": 1.9458, "step": 27 }, { "epoch": 0.07061790668348046, "grad_norm": 9.101158142089844, "learning_rate": 9.721166032953105e-06, "loss": 2.0452, "step": 28 }, { "epoch": 0.07313997477931904, "grad_norm": 9.63065242767334, "learning_rate": 9.708491761723701e-06, "loss": 1.9648, "step": 29 }, { "epoch": 0.07566204287515763, "grad_norm": 9.038743019104004, "learning_rate": 9.695817490494297e-06, "loss": 1.9469, "step": 30 }, { "epoch": 0.07818411097099622, "grad_norm": 9.375469207763672, "learning_rate": 9.683143219264893e-06, "loss": 2.0387, "step": 31 }, { "epoch": 0.0807061790668348, "grad_norm": 9.170827865600586, "learning_rate": 9.670468948035488e-06, "loss": 2.0396, "step": 32 }, { "epoch": 0.0832282471626734, "grad_norm": 9.155048370361328, "learning_rate": 9.657794676806086e-06, "loss": 2.1346, "step": 33 }, { "epoch": 0.08575031525851198, "grad_norm": 9.660133361816406, "learning_rate": 9.64512040557668e-06, "loss": 1.814, "step": 34 }, { "epoch": 0.08827238335435057, "grad_norm": 9.577939987182617, "learning_rate": 9.632446134347276e-06, "loss": 1.9353, "step": 35 }, { "epoch": 0.09079445145018916, "grad_norm": 9.347375869750977, "learning_rate": 9.619771863117872e-06, "loss": 1.8748, "step": 36 }, { "epoch": 0.09331651954602774, "grad_norm": 9.635443687438965, "learning_rate": 9.607097591888467e-06, "loss": 2.0299, "step": 37 }, { "epoch": 0.09583858764186633, "grad_norm": 9.470548629760742, "learning_rate": 9.594423320659063e-06, "loss": 1.966, "step": 38 }, { "epoch": 0.09836065573770492, "grad_norm": 9.292247772216797, "learning_rate": 9.581749049429659e-06, "loss": 1.8779, "step": 39 }, { "epoch": 0.1008827238335435, "grad_norm": 9.822443962097168, "learning_rate": 9.569074778200255e-06, "loss": 1.8484, "step": 40 }, { "epoch": 0.1034047919293821, "grad_norm": 8.519198417663574, "learning_rate": 9.55640050697085e-06, "loss": 1.7622, "step": 41 }, { "epoch": 0.10592686002522068, "grad_norm": 9.220460891723633, "learning_rate": 9.543726235741445e-06, "loss": 1.8659, "step": 42 }, { "epoch": 0.10844892812105927, "grad_norm": 8.564704895019531, "learning_rate": 9.531051964512042e-06, "loss": 1.8478, "step": 43 }, { "epoch": 0.11097099621689786, "grad_norm": 9.696242332458496, "learning_rate": 9.518377693282636e-06, "loss": 1.8625, "step": 44 }, { "epoch": 0.11349306431273644, "grad_norm": 9.121529579162598, "learning_rate": 9.505703422053234e-06, "loss": 1.9736, "step": 45 }, { "epoch": 0.11601513240857503, "grad_norm": 9.0239896774292, "learning_rate": 9.493029150823828e-06, "loss": 1.8687, "step": 46 }, { "epoch": 0.11853720050441362, "grad_norm": 8.566268920898438, "learning_rate": 9.480354879594424e-06, "loss": 1.7855, "step": 47 }, { "epoch": 0.1210592686002522, "grad_norm": 9.728326797485352, "learning_rate": 9.46768060836502e-06, "loss": 2.0394, "step": 48 }, { "epoch": 0.1235813366960908, "grad_norm": 8.644201278686523, "learning_rate": 9.455006337135616e-06, "loss": 1.8265, "step": 49 }, { "epoch": 0.12610340479192939, "grad_norm": 9.00877857208252, "learning_rate": 9.442332065906211e-06, "loss": 1.9655, "step": 50 }, { "epoch": 0.12610340479192939, "eval_loss": 1.8593257665634155, "eval_runtime": 19.3972, "eval_samples_per_second": 36.345, "eval_steps_per_second": 18.198, "step": 50 } ], "logging_steps": 1, "max_steps": 794, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 267147834912768.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }