{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.10706638115631692, "eval_steps": 100, "global_step": 200, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 447.8, "completions/max_terminated_length": 444.8, "completions/mean_length": 301.325, "completions/mean_terminated_length": 299.0325012207031, "completions/min_length": 185.2, "completions/min_terminated_length": 185.2, "entropy": 0.3038703538477421, "epoch": 0.0026766595289079227, "frac_reward_zero_std": 0.6, "grad_norm": 0.0016536490293219686, "learning_rate": 4.0000000000000003e-07, "loss": 0.009808334708213805, "num_tokens": 33162.0, "reward": 0.793749988079071, "reward_std": 0.16348075568675996, "rewards/math_reward_func/mean": 0.793749988079071, "rewards/math_reward_func/std": 0.29990293383598327, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 391.0, "completions/max_terminated_length": 387.0, "completions/mean_length": 281.2, "completions/mean_terminated_length": 276.5589294433594, "completions/min_length": 191.4, "completions/min_terminated_length": 191.4, "entropy": 0.27801988199353217, "epoch": 0.0053533190578158455, "frac_reward_zero_std": 0.75, "grad_norm": 0.002916519995778799, "learning_rate": 9.000000000000001e-07, "loss": 0.012077806890010834, "num_tokens": 64086.0, "reward": 0.9200000047683716, "reward_std": 0.1209807574748993, "rewards/math_reward_func/mean": 0.9200000047683716, "rewards/math_reward_func/std": 0.23296340107917785, "step": 10 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 397.4, "completions/max_terminated_length": 397.4, "completions/mean_length": 285.4125, "completions/mean_terminated_length": 285.4125, "completions/min_length": 188.2, "completions/min_terminated_length": 188.2, "entropy": 0.2916003692895174, "epoch": 0.008029978586723769, "frac_reward_zero_std": 0.7, "grad_norm": 0.001736775622703135, "learning_rate": 1.4000000000000001e-06, "loss": 0.006353498995304107, "num_tokens": 95279.0, "reward": 0.8412499785423279, "reward_std": 0.14291359782218932, "rewards/math_reward_func/mean": 0.8412499785423279, "rewards/math_reward_func/std": 0.3423957586288452, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 403.8, "completions/max_terminated_length": 403.8, "completions/mean_length": 286.3625, "completions/mean_terminated_length": 286.3625, "completions/min_length": 198.4, "completions/min_terminated_length": 198.4, "entropy": 0.2887205803766847, "epoch": 0.010706638115631691, "frac_reward_zero_std": 0.55, "grad_norm": 0.0028273477219045162, "learning_rate": 1.9000000000000002e-06, "loss": 0.018129627406597137, "num_tokens": 127736.0, "reward": 0.8524999856948853, "reward_std": 0.21892304420471193, "rewards/math_reward_func/mean": 0.8524999856948853, "rewards/math_reward_func/std": 0.3371007859706879, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 416.6, "completions/max_terminated_length": 416.6, "completions/mean_length": 280.6875, "completions/mean_terminated_length": 280.6875, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.31409402918070556, "epoch": 0.013383297644539615, "frac_reward_zero_std": 0.75, "grad_norm": 0.0020543853752315044, "learning_rate": 2.4000000000000003e-06, "loss": -0.0009689867496490478, "num_tokens": 159155.0, "reward": 0.9075000047683716, "reward_std": 0.09288674890995026, "rewards/math_reward_func/mean": 0.9075000047683716, "rewards/math_reward_func/std": 0.19665745496749878, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 476.6, "completions/max_terminated_length": 469.0, "completions/mean_length": 338.575, "completions/mean_terminated_length": 336.1683349609375, "completions/min_length": 221.6, "completions/min_terminated_length": 221.6, "entropy": 0.2726387668401003, "epoch": 0.016059957173447537, "frac_reward_zero_std": 0.4, "grad_norm": 0.002868631621822715, "learning_rate": 2.9e-06, "loss": 0.015286873281002044, "num_tokens": 196041.0, "reward": 0.7837499856948853, "reward_std": 0.23348075151443481, "rewards/math_reward_func/mean": 0.7837499856948853, "rewards/math_reward_func/std": 0.3736962258815765, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 454.2, "completions/max_terminated_length": 445.4, "completions/mean_length": 297.225, "completions/mean_terminated_length": 292.10178833007814, "completions/min_length": 181.6, "completions/min_terminated_length": 181.6, "entropy": 0.2940476995892823, "epoch": 0.01873661670235546, "frac_reward_zero_std": 0.55, "grad_norm": 0.0014004572294652462, "learning_rate": 3.4000000000000005e-06, "loss": 0.001631837710738182, "num_tokens": 228647.0, "reward": 0.83125, "reward_std": 0.20946152210235597, "rewards/math_reward_func/mean": 0.83125, "rewards/math_reward_func/std": 0.3457089066505432, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05, "completions/max_length": 433.0, "completions/max_terminated_length": 413.6, "completions/mean_length": 282.0, "completions/mean_terminated_length": 272.3958374023438, "completions/min_length": 171.4, "completions/min_terminated_length": 171.4, "entropy": 0.29167091082781554, "epoch": 0.021413276231263382, "frac_reward_zero_std": 0.7, "grad_norm": 0.0027562258765101433, "learning_rate": 3.900000000000001e-06, "loss": 0.013593007624149323, "num_tokens": 260087.0, "reward": 0.8049999713897705, "reward_std": 0.14121417403221131, "rewards/math_reward_func/mean": 0.8049999713897705, "rewards/math_reward_func/std": 0.32876750230789187, "step": 40 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 420.4, "completions/max_terminated_length": 420.4, "completions/mean_length": 283.45, "completions/mean_terminated_length": 283.45, "completions/min_length": 155.2, "completions/min_terminated_length": 155.2, "entropy": 0.28917145598679783, "epoch": 0.024089935760171308, "frac_reward_zero_std": 0.7, "grad_norm": 0.0018424366135150194, "learning_rate": 4.4e-06, "loss": 0.03561306893825531, "num_tokens": 291523.0, "reward": 0.8875, "reward_std": 0.13499999642372132, "rewards/math_reward_func/mean": 0.8875, "rewards/math_reward_func/std": 0.2541318416595459, "step": 45 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 401.4, "completions/max_terminated_length": 401.4, "completions/mean_length": 259.8375, "completions/mean_terminated_length": 259.8375, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.3092456264421344, "epoch": 0.02676659528907923, "frac_reward_zero_std": 0.7, "grad_norm": 0.003556865965947509, "learning_rate": 4.9000000000000005e-06, "loss": -0.005310375988483429, "num_tokens": 320942.0, "reward": 0.8737500071525574, "reward_std": 0.14095207452774047, "rewards/math_reward_func/mean": 0.8737500071525574, "rewards/math_reward_func/std": 0.18471888303756714, "step": 50 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 412.4, "completions/max_terminated_length": 412.4, "completions/mean_length": 271.425, "completions/mean_terminated_length": 271.425, "completions/min_length": 177.8, "completions/min_terminated_length": 177.8, "entropy": 0.2713283909484744, "epoch": 0.029443254817987152, "frac_reward_zero_std": 0.7, "grad_norm": 0.0014988939510658383, "learning_rate": 4.994574064026045e-06, "loss": 0.009379406273365021, "num_tokens": 351420.0, "reward": 0.8512500047683715, "reward_std": 0.09848076105117798, "rewards/math_reward_func/mean": 0.8512500047683715, "rewards/math_reward_func/std": 0.26999210715293886, "step": 55 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 416.4, "completions/max_terminated_length": 412.6, "completions/mean_length": 280.4625, "completions/mean_terminated_length": 278.21083984375, "completions/min_length": 157.2, "completions/min_terminated_length": 157.2, "entropy": 0.3262932809069753, "epoch": 0.032119914346895075, "frac_reward_zero_std": 0.65, "grad_norm": 0.002003510482609272, "learning_rate": 4.987791644058601e-06, "loss": -3.32362949848175e-05, "num_tokens": 382941.0, "reward": 0.81875, "reward_std": 0.14446152094751596, "rewards/math_reward_func/mean": 0.81875, "rewards/math_reward_func/std": 0.3091198801994324, "step": 60 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 423.4, "completions/max_terminated_length": 423.4, "completions/mean_length": 285.35, "completions/mean_terminated_length": 285.35, "completions/min_length": 195.8, "completions/min_terminated_length": 195.8, "entropy": 0.3003417044878006, "epoch": 0.034796573875803, "frac_reward_zero_std": 0.75, "grad_norm": 0.0022777975536882877, "learning_rate": 4.981009224091156e-06, "loss": -0.0026859112083911897, "num_tokens": 414565.0, "reward": 0.943750011920929, "reward_std": 0.11249999701976776, "rewards/math_reward_func/mean": 0.943750011920929, "rewards/math_reward_func/std": 0.19648169875144958, "step": 65 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 427.2, "completions/max_terminated_length": 427.2, "completions/mean_length": 266.75, "completions/mean_terminated_length": 266.75, "completions/min_length": 160.8, "completions/min_terminated_length": 160.8, "entropy": 0.27976486682891843, "epoch": 0.03747323340471092, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 4.974226804123712e-06, "loss": 0.0024851024150848388, "num_tokens": 444505.0, "reward": 0.9200000047683716, "reward_std": 0.06999999824911356, "rewards/math_reward_func/mean": 0.9200000047683716, "rewards/math_reward_func/std": 0.18933700323104857, "step": 70 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.2, "completions/max_terminated_length": 367.2, "completions/mean_length": 256.35, "completions/mean_terminated_length": 256.35, "completions/min_length": 179.8, "completions/min_terminated_length": 179.8, "entropy": 0.2776689689606428, "epoch": 0.04014989293361884, "frac_reward_zero_std": 0.7, "grad_norm": 0.0, "learning_rate": 4.9674443841562676e-06, "loss": 0.0062458813190460205, "num_tokens": 473165.0, "reward": 0.9212500095367432, "reward_std": 0.13848076164722442, "rewards/math_reward_func/mean": 0.9212500095367432, "rewards/math_reward_func/std": 0.2154984414577484, "step": 75 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 390.4, "completions/max_terminated_length": 390.4, "completions/mean_length": 266.0125, "completions/mean_terminated_length": 266.0125, "completions/min_length": 170.2, "completions/min_terminated_length": 170.2, "entropy": 0.29570526592433455, "epoch": 0.042826552462526764, "frac_reward_zero_std": 0.9, "grad_norm": 0.002651757327839732, "learning_rate": 4.9606619641888234e-06, "loss": 0.0120128333568573, "num_tokens": 503946.0, "reward": 0.9662500023841858, "reward_std": 0.048480761051177976, "rewards/math_reward_func/mean": 0.9662500023841858, "rewards/math_reward_func/std": 0.10648170113563538, "step": 80 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 437.6, "completions/max_terminated_length": 437.6, "completions/mean_length": 305.625, "completions/mean_terminated_length": 303.7883361816406, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.277633311599493, "epoch": 0.045503211991434686, "frac_reward_zero_std": 0.8, "grad_norm": 0.0018970033852383494, "learning_rate": 4.9538795442213784e-06, "loss": 0.01977337896823883, "num_tokens": 537420.0, "reward": 0.8637500047683716, "reward_std": 0.09095207750797271, "rewards/math_reward_func/mean": 0.8637500047683716, "rewards/math_reward_func/std": 0.27909453511238097, "step": 85 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 424.2, "completions/max_terminated_length": 413.2, "completions/mean_length": 279.0125, "completions/mean_terminated_length": 276.7100036621094, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.27373425792902706, "epoch": 0.048179871520342615, "frac_reward_zero_std": 0.7, "grad_norm": 0.002039178041741252, "learning_rate": 4.947097124253934e-06, "loss": 0.009770887345075608, "num_tokens": 568929.0, "reward": 0.8537499666213989, "reward_std": 0.14544228315353394, "rewards/math_reward_func/mean": 0.8537499666213989, "rewards/math_reward_func/std": 0.2981793940067291, "step": 90 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 405.8, "completions/max_terminated_length": 399.2, "completions/mean_length": 271.3875, "completions/mean_terminated_length": 269.21917114257815, "completions/min_length": 173.6, "completions/min_terminated_length": 173.6, "entropy": 0.2791801610961556, "epoch": 0.05085653104925054, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.940314704286489e-06, "loss": 0.007668537646532058, "num_tokens": 599044.0, "reward": 0.9775000095367432, "reward_std": 0.0449999988079071, "rewards/math_reward_func/mean": 0.9775000095367432, "rewards/math_reward_func/std": 0.0899999976158142, "step": 95 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 370.2, "completions/max_terminated_length": 370.2, "completions/mean_length": 279.225, "completions/mean_terminated_length": 279.225, "completions/min_length": 187.4, "completions/min_terminated_length": 187.4, "entropy": 0.26580147063359616, "epoch": 0.05353319057815846, "frac_reward_zero_std": 0.75, "grad_norm": 0.0, "learning_rate": 4.933532284319045e-06, "loss": 0.011092402786016465, "num_tokens": 630158.0, "reward": 0.8524999856948853, "reward_std": 0.09598076045513153, "rewards/math_reward_func/mean": 0.8524999856948853, "rewards/math_reward_func/std": 0.2916661620140076, "step": 100 }, { "epoch": 0.05353319057815846, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.015, "eval_completions/max_length": 328.59, "eval_completions/max_terminated_length": 323.68, "eval_completions/mean_length": 281.8625, "eval_completions/mean_terminated_length": 279.78166748046874, "eval_completions/min_length": 238.22, "eval_completions/min_terminated_length": 238.22, "eval_entropy": 0.30293387338519095, "eval_frac_reward_zero_std": 0.79, "eval_loss": 0.006132281851023436, "eval_num_tokens": 630158.0, "eval_reward": 0.9024999978393317, "eval_reward_std": 0.09816543877124786, "eval_rewards/math_reward_func/mean": 0.9024999978393317, "eval_rewards/math_reward_func/std": 0.09816543936729431, "eval_runtime": 2872.7231, "eval_samples_per_second": 0.035, "eval_steps_per_second": 0.009, "step": 100 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 393.8, "completions/max_terminated_length": 393.8, "completions/mean_length": 276.225, "completions/mean_terminated_length": 276.225, "completions/min_length": 189.6, "completions/min_terminated_length": 189.6, "entropy": 0.27943384218961, "epoch": 0.05620985010706638, "frac_reward_zero_std": 0.85, "grad_norm": 0.0, "learning_rate": 4.926749864351601e-06, "loss": 0.014298515021800995, "num_tokens": 660728.0, "reward": 0.9650000095367431, "reward_std": 0.06999999880790711, "rewards/math_reward_func/mean": 0.9650000095367431, "rewards/math_reward_func/std": 0.1100000023841858, "step": 105 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0375, "completions/max_length": 445.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 287.45, "completions/mean_terminated_length": 279.3376220703125, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.27698346730321644, "epoch": 0.058886509635974305, "frac_reward_zero_std": 0.65, "grad_norm": 0.004079123958945274, "learning_rate": 4.919967444384157e-06, "loss": 0.0098875492811203, "num_tokens": 693020.0, "reward": 0.7924999833106995, "reward_std": 0.12386751137673854, "rewards/math_reward_func/mean": 0.7924999833106995, "rewards/math_reward_func/std": 0.3255069851875305, "step": 110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 438.2, "completions/max_terminated_length": 418.8, "completions/mean_length": 289.2375, "completions/mean_terminated_length": 286.461669921875, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.26555775087326766, "epoch": 0.06156316916488223, "frac_reward_zero_std": 0.8, "grad_norm": 0.0, "learning_rate": 4.913185024416712e-06, "loss": 0.005856743454933167, "num_tokens": 724727.0, "reward": 0.8762500047683716, "reward_std": 0.09348075985908508, "rewards/math_reward_func/mean": 0.8762500047683716, "rewards/math_reward_func/std": 0.3045404672622681, "step": 115 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05, "completions/max_length": 410.2, "completions/max_terminated_length": 386.4, "completions/mean_length": 302.8375, "completions/mean_terminated_length": 293.7517883300781, "completions/min_length": 203.8, "completions/min_terminated_length": 203.8, "entropy": 0.26143238730728624, "epoch": 0.06423982869379015, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.906402604449268e-06, "loss": 0.019596871733665467, "num_tokens": 757854.0, "reward": 0.9087499976158142, "reward_std": 0.053480759263038635, "rewards/math_reward_func/mean": 0.9087499976158142, "rewards/math_reward_func/std": 0.20698015093803407, "step": 120 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 435.0, "completions/max_terminated_length": 420.6, "completions/mean_length": 292.5625, "completions/mean_terminated_length": 289.5350036621094, "completions/min_length": 190.4, "completions/min_terminated_length": 190.4, "entropy": 0.25466044507920743, "epoch": 0.06691648822269808, "frac_reward_zero_std": 0.85, "grad_norm": 0.0, "learning_rate": 4.899620184481824e-06, "loss": 0.00965617150068283, "num_tokens": 790475.0, "reward": 0.9437499880790711, "reward_std": 0.07446152269840241, "rewards/math_reward_func/mean": 0.9437499880790711, "rewards/math_reward_func/std": 0.13404201865196227, "step": 125 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 389.6, "completions/max_terminated_length": 389.6, "completions/mean_length": 278.55, "completions/mean_terminated_length": 278.55, "completions/min_length": 185.6, "completions/min_terminated_length": 185.6, "entropy": 0.29498828779906033, "epoch": 0.069593147751606, "frac_reward_zero_std": 0.95, "grad_norm": 0.0, "learning_rate": 4.8928377645143796e-06, "loss": 0.007200612872838974, "num_tokens": 822595.0, "reward": 0.9887500047683716, "reward_std": 0.02249999940395355, "rewards/math_reward_func/mean": 0.9887500047683716, "rewards/math_reward_func/std": 0.0449999988079071, "step": 130 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 367.2, "completions/max_terminated_length": 367.2, "completions/mean_length": 270.925, "completions/mean_terminated_length": 270.925, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.2816799750551581, "epoch": 0.07226980728051392, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.886055344546935e-06, "loss": 0.0, "num_tokens": 853133.0, "reward": 0.9549999952316284, "reward_std": 0.0, "rewards/math_reward_func/mean": 0.9549999952316284, "rewards/math_reward_func/std": 0.0804984450340271, "step": 135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 396.4, "completions/max_terminated_length": 396.4, "completions/mean_length": 256.5875, "completions/mean_terminated_length": 256.5875, "completions/min_length": 174.6, "completions/min_terminated_length": 174.6, "entropy": 0.2529933013021946, "epoch": 0.07494646680942184, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.8792729245794905e-06, "loss": 0.0, "num_tokens": 881936.0, "reward": 0.9549999952316284, "reward_std": 0.0, "rewards/math_reward_func/mean": 0.9549999952316284, "rewards/math_reward_func/std": 0.0804984450340271, "step": 140 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 416.4, "completions/max_terminated_length": 394.0, "completions/mean_length": 271.2625, "completions/mean_terminated_length": 268.5325012207031, "completions/min_length": 177.4, "completions/min_terminated_length": 177.4, "entropy": 0.2719669111073017, "epoch": 0.07762312633832977, "frac_reward_zero_std": 0.95, "grad_norm": 0.0, "learning_rate": 4.8724905046120455e-06, "loss": -0.0016425080597400666, "num_tokens": 912137.0, "reward": 0.9887500047683716, "reward_std": 0.02249999940395355, "rewards/math_reward_func/mean": 0.9887500047683716, "rewards/math_reward_func/std": 0.0449999988079071, "step": 145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 354.6, "completions/max_terminated_length": 354.6, "completions/mean_length": 249.725, "completions/mean_terminated_length": 249.725, "completions/min_length": 152.4, "completions/min_terminated_length": 152.4, "entropy": 0.28879304584115745, "epoch": 0.08029978586723768, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.865708084644601e-06, "loss": -0.005524076521396637, "num_tokens": 940467.0, "reward": 0.9537500023841858, "reward_std": 0.045952077209949496, "rewards/math_reward_func/mean": 0.9537500023841858, "rewards/math_reward_func/std": 0.12036577463150025, "step": 150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 393.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 278.3875, "completions/mean_terminated_length": 275.22083740234376, "completions/min_length": 183.2, "completions/min_terminated_length": 183.2, "entropy": 0.2648607546463609, "epoch": 0.08297644539614561, "frac_reward_zero_std": 0.95, "grad_norm": 0.0012877928093075752, "learning_rate": 4.858925664677157e-06, "loss": 0.0052356265485286714, "num_tokens": 971870.0, "reward": 0.94375, "reward_std": 0.02249999940395355, "rewards/math_reward_func/mean": 0.94375, "rewards/math_reward_func/std": 0.1254984438419342, "step": 155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 366.8, "completions/max_terminated_length": 366.8, "completions/mean_length": 251.9625, "completions/mean_terminated_length": 251.9625, "completions/min_length": 167.6, "completions/min_terminated_length": 167.6, "entropy": 0.28362263944000005, "epoch": 0.08565310492505353, "frac_reward_zero_std": 0.95, "grad_norm": 0.0, "learning_rate": 4.852143244709713e-06, "loss": 0.003054547682404518, "num_tokens": 1000747.0, "reward": 0.9887500047683716, "reward_std": 0.02249999940395355, "rewards/math_reward_func/mean": 0.9887500047683716, "rewards/math_reward_func/std": 0.0449999988079071, "step": 160 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0125, "completions/max_length": 431.0, "completions/max_terminated_length": 421.2, "completions/mean_length": 279.0375, "completions/mean_terminated_length": 275.7808349609375, "completions/min_length": 158.8, "completions/min_terminated_length": 158.8, "entropy": 0.3094722624868155, "epoch": 0.08832976445396146, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.845360824742268e-06, "loss": 0.01083858534693718, "num_tokens": 1031886.0, "reward": 0.9549999952316284, "reward_std": 0.05196152329444885, "rewards/math_reward_func/mean": 0.9549999952316284, "rewards/math_reward_func/std": 0.12296340465545655, "step": 165 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.05, "completions/max_length": 417.0, "completions/max_terminated_length": 398.8, "completions/mean_length": 269.45, "completions/mean_terminated_length": 256.3666687011719, "completions/min_length": 159.6, "completions/min_terminated_length": 159.6, "entropy": 0.2650467408820987, "epoch": 0.09100642398286937, "frac_reward_zero_std": 0.95, "grad_norm": 0.0, "learning_rate": 4.838578404774824e-06, "loss": -0.007036852836608887, "num_tokens": 1062198.0, "reward": 0.94375, "reward_std": 0.02249999940395355, "rewards/math_reward_func/mean": 0.94375, "rewards/math_reward_func/std": 0.1254984438419342, "step": 170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 380.8, "completions/max_terminated_length": 380.8, "completions/mean_length": 272.1625, "completions/mean_terminated_length": 272.1625, "completions/min_length": 183.4, "completions/min_terminated_length": 183.4, "entropy": 0.2562641527503729, "epoch": 0.0936830835117773, "frac_reward_zero_std": 0.9, "grad_norm": 0.0015640855999663472, "learning_rate": 4.83179598480738e-06, "loss": 0.0044931560754776, "num_tokens": 1092843.0, "reward": 0.94375, "reward_std": 0.048480761051177976, "rewards/math_reward_func/mean": 0.94375, "rewards/math_reward_func/std": 0.13404202461242676, "step": 175 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 404.6, "completions/max_terminated_length": 404.6, "completions/mean_length": 274.4, "completions/mean_terminated_length": 274.4, "completions/min_length": 171.6, "completions/min_terminated_length": 171.6, "entropy": 0.2781650984659791, "epoch": 0.09635974304068523, "frac_reward_zero_std": 0.9, "grad_norm": 0.0, "learning_rate": 4.825013564839936e-06, "loss": 0.010695323348045349, "num_tokens": 1124063.0, "reward": 0.9200000047683716, "reward_std": 0.04999999701976776, "rewards/math_reward_func/mean": 0.9200000047683716, "rewards/math_reward_func/std": 0.1367877960205078, "step": 180 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 387.6, "completions/max_terminated_length": 387.6, "completions/mean_length": 267.95, "completions/mean_terminated_length": 267.95, "completions/min_length": 166.2, "completions/min_terminated_length": 166.2, "entropy": 0.26084629725664854, "epoch": 0.09903640256959315, "frac_reward_zero_std": 0.95, "grad_norm": 0.0, "learning_rate": 4.818231144872491e-06, "loss": -0.0019785046577453615, "num_tokens": 1154747.0, "reward": 0.9537500023841858, "reward_std": 0.002500000037252903, "rewards/math_reward_func/mean": 0.9537500023841858, "rewards/math_reward_func/std": 0.08285530209541321, "step": 185 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 425.2, "completions/max_terminated_length": 419.0, "completions/mean_length": 291.6625, "completions/mean_terminated_length": 287.6910766601562, "completions/min_length": 192.2, "completions/min_terminated_length": 192.2, "entropy": 0.28378621079027655, "epoch": 0.10171306209850108, "frac_reward_zero_std": 0.85, "grad_norm": 0.0018047435441985726, "learning_rate": 4.811448724905047e-06, "loss": 0.006797170639038086, "num_tokens": 1187208.0, "reward": 0.8274999976158142, "reward_std": 0.07830035388469696, "rewards/math_reward_func/mean": 0.8274999976158142, "rewards/math_reward_func/std": 0.24652022123336792, "step": 190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.025, "completions/max_length": 432.8, "completions/max_terminated_length": 426.6, "completions/mean_length": 306.4875, "completions/mean_terminated_length": 301.7196472167969, "completions/min_length": 192.4, "completions/min_terminated_length": 192.4, "entropy": 0.28082586638629436, "epoch": 0.10438972162740899, "frac_reward_zero_std": 0.85, "grad_norm": 0.0, "learning_rate": 4.804666304937602e-06, "loss": 0.004231558740139007, "num_tokens": 1221083.0, "reward": 0.8625, "reward_std": 0.0728849157691002, "rewards/math_reward_func/mean": 0.8625, "rewards/math_reward_func/std": 0.1820126235485077, "step": 195 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 400.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 270.775, "completions/mean_terminated_length": 270.775, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.2751172995194793, "epoch": 0.10706638115631692, "frac_reward_zero_std": 0.9, "grad_norm": 0.002758933464065194, "learning_rate": 4.7978838849701575e-06, "loss": -3.464482724666595e-05, "num_tokens": 1251405.0, "reward": 0.9550000071525574, "reward_std": 0.0449999988079071, "rewards/math_reward_func/mean": 0.9550000071525574, "rewards/math_reward_func/std": 0.11756032109260559, "step": 200 }, { "epoch": 0.10706638115631692, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0125, "eval_completions/max_length": 330.31, "eval_completions/max_terminated_length": 323.86, "eval_completions/mean_length": 283.21, "eval_completions/mean_terminated_length": 281.17250061035156, "eval_completions/min_length": 237.57, "eval_completions/min_terminated_length": 237.57, "eval_entropy": 0.30089579612016676, "eval_frac_reward_zero_std": 0.86, "eval_loss": 0.006860610097646713, "eval_num_tokens": 1251405.0, "eval_reward": 0.9137499988824129, "eval_reward_std": 0.06005241829901933, "eval_rewards/math_reward_func/mean": 0.9137499988824129, "eval_rewards/math_reward_func/std": 0.06005241859704256, "eval_runtime": 2890.1242, "eval_samples_per_second": 0.035, "eval_steps_per_second": 0.009, "step": 200 } ], "logging_steps": 5, "max_steps": 3736, "num_input_tokens_seen": 1251405, "num_train_epochs": 2, "save_steps": 25, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }