{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.0354735721887194, "eval_steps": 500, "global_step": 50, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.000709471443774388, "grad_norm": 265.86757613893764, "learning_rate": 0.0, "loss": 2.1209, "step": 1 }, { "epoch": 0.001418942887548776, "grad_norm": 256.9364781809131, "learning_rate": 2.1276595744680852e-07, "loss": 1.9508, "step": 2 }, { "epoch": 0.002128414331323164, "grad_norm": 261.72185657120684, "learning_rate": 4.2553191489361704e-07, "loss": 2.1995, "step": 3 }, { "epoch": 0.002837885775097552, "grad_norm": 261.0961921742672, "learning_rate": 6.382978723404255e-07, "loss": 2.0081, "step": 4 }, { "epoch": 0.0035473572188719402, "grad_norm": 255.1019965019492, "learning_rate": 8.510638297872341e-07, "loss": 2.0828, "step": 5 }, { "epoch": 0.004256828662646328, "grad_norm": 201.9040896756704, "learning_rate": 1.0638297872340427e-06, "loss": 1.6403, "step": 6 }, { "epoch": 0.004966300106420717, "grad_norm": 198.32057240838185, "learning_rate": 1.276595744680851e-06, "loss": 1.6037, "step": 7 }, { "epoch": 0.005675771550195104, "grad_norm": 91.64675018232528, "learning_rate": 1.4893617021276596e-06, "loss": 1.0155, "step": 8 }, { "epoch": 0.006385242993969493, "grad_norm": 71.73605818031255, "learning_rate": 1.7021276595744682e-06, "loss": 0.8153, "step": 9 }, { "epoch": 0.0070947144377438804, "grad_norm": 24.941283646351614, "learning_rate": 1.9148936170212763e-06, "loss": 0.3811, "step": 10 }, { "epoch": 0.007804185881518269, "grad_norm": 31.071639133351898, "learning_rate": 2.1276595744680853e-06, "loss": 0.3851, "step": 11 }, { "epoch": 0.008513657325292657, "grad_norm": 20.77631646223411, "learning_rate": 2.340425531914894e-06, "loss": 0.2891, "step": 12 }, { "epoch": 0.009223128769067045, "grad_norm": 13.477671793796729, "learning_rate": 2.553191489361702e-06, "loss": 0.2528, "step": 13 }, { "epoch": 0.009932600212841433, "grad_norm": 2.6555809333853024, "learning_rate": 2.7659574468085106e-06, "loss": 0.1542, "step": 14 }, { "epoch": 0.010642071656615822, "grad_norm": 2.898054397668401, "learning_rate": 2.978723404255319e-06, "loss": 0.1762, "step": 15 }, { "epoch": 0.011351543100390209, "grad_norm": 2.1504034279865385, "learning_rate": 3.1914893617021277e-06, "loss": 0.1403, "step": 16 }, { "epoch": 0.012061014544164597, "grad_norm": 5.259173556940105, "learning_rate": 3.4042553191489363e-06, "loss": 0.1561, "step": 17 }, { "epoch": 0.012770485987938986, "grad_norm": 4.169332828012237, "learning_rate": 3.6170212765957445e-06, "loss": 0.1632, "step": 18 }, { "epoch": 0.013479957431713374, "grad_norm": 1.9000129489744462, "learning_rate": 3.829787234042553e-06, "loss": 0.1667, "step": 19 }, { "epoch": 0.014189428875487761, "grad_norm": 7.769485079957688, "learning_rate": 4.042553191489362e-06, "loss": 0.1893, "step": 20 }, { "epoch": 0.01489890031926215, "grad_norm": 11.504242044440007, "learning_rate": 4.255319148936171e-06, "loss": 0.2085, "step": 21 }, { "epoch": 0.015608371763036538, "grad_norm": 7.145142567820805, "learning_rate": 4.468085106382979e-06, "loss": 0.1913, "step": 22 }, { "epoch": 0.016317843206810925, "grad_norm": 2.736186486725912, "learning_rate": 4.680851063829788e-06, "loss": 0.1741, "step": 23 }, { "epoch": 0.017027314650585313, "grad_norm": 5.7916893585600695, "learning_rate": 4.893617021276596e-06, "loss": 0.1408, "step": 24 }, { "epoch": 0.0177367860943597, "grad_norm": 4.68647394202743, "learning_rate": 5.106382978723404e-06, "loss": 0.1632, "step": 25 }, { "epoch": 0.01844625753813409, "grad_norm": 2.0524513403445757, "learning_rate": 5.319148936170213e-06, "loss": 0.1563, "step": 26 }, { "epoch": 0.01915572898190848, "grad_norm": 1.588252902269814, "learning_rate": 5.531914893617021e-06, "loss": 0.1582, "step": 27 }, { "epoch": 0.019865200425682867, "grad_norm": 3.5755825255555225, "learning_rate": 5.74468085106383e-06, "loss": 0.1451, "step": 28 }, { "epoch": 0.020574671869457255, "grad_norm": 7.715437458161659, "learning_rate": 5.957446808510638e-06, "loss": 0.157, "step": 29 }, { "epoch": 0.021284143313231644, "grad_norm": 19.70487906339096, "learning_rate": 6.1702127659574465e-06, "loss": 0.1725, "step": 30 }, { "epoch": 0.02199361475700603, "grad_norm": 2.21235512385039, "learning_rate": 6.3829787234042555e-06, "loss": 0.1318, "step": 31 }, { "epoch": 0.022703086200780417, "grad_norm": 5.021653379215672, "learning_rate": 6.5957446808510645e-06, "loss": 0.1795, "step": 32 }, { "epoch": 0.023412557644554806, "grad_norm": 25.495061679606632, "learning_rate": 6.808510638297873e-06, "loss": 0.1586, "step": 33 }, { "epoch": 0.024122029088329194, "grad_norm": 4.7481186730757035, "learning_rate": 7.021276595744681e-06, "loss": 0.1552, "step": 34 }, { "epoch": 0.024831500532103583, "grad_norm": 3.6399107060391365, "learning_rate": 7.234042553191489e-06, "loss": 0.1573, "step": 35 }, { "epoch": 0.02554097197587797, "grad_norm": 3.3754400837055707, "learning_rate": 7.446808510638298e-06, "loss": 0.149, "step": 36 }, { "epoch": 0.02625044341965236, "grad_norm": 2.8787349588257842, "learning_rate": 7.659574468085105e-06, "loss": 0.154, "step": 37 }, { "epoch": 0.026959914863426748, "grad_norm": 2.7404452323712416, "learning_rate": 7.872340425531914e-06, "loss": 0.1563, "step": 38 }, { "epoch": 0.027669386307201137, "grad_norm": 1.8896448248412065, "learning_rate": 8.085106382978723e-06, "loss": 0.1568, "step": 39 }, { "epoch": 0.028378857750975522, "grad_norm": 6.6357403507533395, "learning_rate": 8.297872340425532e-06, "loss": 0.1608, "step": 40 }, { "epoch": 0.02908832919474991, "grad_norm": 5.2141592606599545, "learning_rate": 8.510638297872341e-06, "loss": 0.1495, "step": 41 }, { "epoch": 0.0297978006385243, "grad_norm": 2.653621690332451, "learning_rate": 8.723404255319149e-06, "loss": 0.1746, "step": 42 }, { "epoch": 0.030507272082298687, "grad_norm": 2.1695013496806483, "learning_rate": 8.936170212765958e-06, "loss": 0.1501, "step": 43 }, { "epoch": 0.031216743526073076, "grad_norm": 1.9610946251979264, "learning_rate": 9.148936170212767e-06, "loss": 0.1614, "step": 44 }, { "epoch": 0.03192621496984746, "grad_norm": 3.806116355988534, "learning_rate": 9.361702127659576e-06, "loss": 0.1649, "step": 45 }, { "epoch": 0.03263568641362185, "grad_norm": 1.6308337200888532, "learning_rate": 9.574468085106385e-06, "loss": 0.1528, "step": 46 }, { "epoch": 0.03334515785739624, "grad_norm": 1.0471741398584304, "learning_rate": 9.787234042553192e-06, "loss": 0.151, "step": 47 }, { "epoch": 0.034054629301170626, "grad_norm": 2.1098801420979885, "learning_rate": 9.999999999999999e-06, "loss": 0.15, "step": 48 }, { "epoch": 0.034764100744945015, "grad_norm": 1.4498470002731845, "learning_rate": 1.0212765957446808e-05, "loss": 0.1385, "step": 49 }, { "epoch": 0.0354735721887194, "grad_norm": 2.4843540113556632, "learning_rate": 1.0425531914893617e-05, "loss": 0.1822, "step": 50 } ], "logging_steps": 1, "max_steps": 2820, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 31813017403392.0, "train_batch_size": 10, "trial_name": null, "trial_params": null }