| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.4705882352941178, |
| "eval_steps": 10, |
| "global_step": 100, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.013888889302810034, |
| "eval_completions/max_length": 1051.047619047619, |
| "eval_completions/max_terminated_length": 869.1190476190476, |
| "eval_completions/mean_length": 481.75398054577056, |
| "eval_completions/mean_terminated_length": 445.6996131170364, |
| "eval_completions/min_length": 214.35714285714286, |
| "eval_completions/min_terminated_length": 214.35714285714286, |
| "eval_entropy": 0.19031873114761852, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 0.0, |
| "eval_reward": 0.4583333432674408, |
| "eval_reward_std": 0.4926568010733241, |
| "eval_rewards/reward_correctness/mean": 0.4583333383003871, |
| "eval_rewards/reward_correctness/std": 0.49265679788021816, |
| "eval_runtime": 621.4365, |
| "eval_samples_per_second": 0.805, |
| "eval_sampling/importance_sampling_ratio/max": 3.0, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9897958558230173, |
| "eval_sampling/importance_sampling_ratio/min": 0.5083668609814984, |
| "eval_sampling/sampling_logp_difference/max": 2.397807226294563, |
| "eval_sampling/sampling_logp_difference/mean": 0.0472170448019391, |
| "eval_steps_per_second": 0.135, |
| "step": 0 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.006510416977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3039.0, |
| "completions/mean_length": 544.44140625, |
| "completions/mean_terminated_length": 527.8781127929688, |
| "completions/min_length": 97.0, |
| "completions/min_terminated_length": 97.0, |
| "entropy": 0.23509776848368347, |
| "epoch": 0.014705882352941176, |
| "frac_reward_zero_std": 0.484375, |
| "grad_norm": 0.09276555104929421, |
| "learning_rate": 0.0, |
| "loss": 0.0089, |
| "num_tokens": 1008798.0, |
| "reward": 0.3372395932674408, |
| "reward_std": 0.4729214012622833, |
| "rewards/reward_correctness/mean": 0.3372395932674408, |
| "rewards/reward_correctness/std": 0.4729214012622833, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9975379109382629, |
| "sampling/importance_sampling_ratio/min": 9.726321137382143e-11, |
| "sampling/sampling_logp_difference/max": 23.053600311279297, |
| "sampling/sampling_logp_difference/mean": 0.042724158614873886, |
| "step": 1, |
| "step_time": 338.8089781026356 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.005859375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3041.0, |
| "completions/mean_length": 531.7975463867188, |
| "completions/mean_terminated_length": 516.8258056640625, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.23647121316753328, |
| "epoch": 0.029411764705882353, |
| "frac_reward_zero_std": 0.4375, |
| "grad_norm": 0.09973641051476721, |
| "learning_rate": 6.000000000000001e-07, |
| "loss": 0.0144, |
| "num_tokens": 1986295.0, |
| "reward": 0.3059895932674408, |
| "reward_std": 0.46097537875175476, |
| "rewards/reward_correctness/mean": 0.3059895932674408, |
| "rewards/reward_correctness/std": 0.46097537875175476, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000666379928589, |
| "sampling/importance_sampling_ratio/min": 0.1762581765651703, |
| "sampling/sampling_logp_difference/max": 1.7358055114746094, |
| "sampling/sampling_logp_difference/mean": 0.015235913917422295, |
| "step": 2, |
| "step_time": 280.3220818070695 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.005859375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2961.0, |
| "completions/mean_length": 541.4909057617188, |
| "completions/mean_terminated_length": 526.5762939453125, |
| "completions/min_length": 93.0, |
| "completions/min_terminated_length": 93.0, |
| "entropy": 0.23211678757797927, |
| "epoch": 0.04411764705882353, |
| "frac_reward_zero_std": 0.453125, |
| "grad_norm": 0.09852825942987486, |
| "learning_rate": 1.2000000000000002e-06, |
| "loss": 0.0094, |
| "num_tokens": 2958029.0, |
| "reward": 0.3697916865348816, |
| "reward_std": 0.4829053580760956, |
| "rewards/reward_correctness/mean": 0.3697916567325592, |
| "rewards/reward_correctness/std": 0.48290541768074036, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999954104423523, |
| "sampling/importance_sampling_ratio/min": 0.1790223866701126, |
| "sampling/sampling_logp_difference/max": 2.3153841495513916, |
| "sampling/sampling_logp_difference/mean": 0.014817245304584503, |
| "step": 3, |
| "step_time": 277.6638293429278 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.005859375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2881.0, |
| "completions/mean_length": 527.78515625, |
| "completions/mean_terminated_length": 512.789794921875, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.2507125422125682, |
| "epoch": 0.058823529411764705, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.09895279498528099, |
| "learning_rate": 1.8e-06, |
| "loss": 0.0037, |
| "num_tokens": 3932327.0, |
| "reward": 0.2903645932674408, |
| "reward_std": 0.45407843589782715, |
| "rewards/reward_correctness/mean": 0.2903645932674408, |
| "rewards/reward_correctness/std": 0.45407843589782715, |
| "sampling/importance_sampling_ratio/max": 2.93314528465271, |
| "sampling/importance_sampling_ratio/mean": 1.0000152587890625, |
| "sampling/importance_sampling_ratio/min": 0.24576407670974731, |
| "sampling/sampling_logp_difference/max": 1.4033832550048828, |
| "sampling/sampling_logp_difference/mean": 0.016350775957107544, |
| "step": 4, |
| "step_time": 275.9684011223726 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3004.0, |
| "completions/mean_length": 557.0267333984375, |
| "completions/mean_terminated_length": 537.2237548828125, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.23487792268861085, |
| "epoch": 0.07352941176470588, |
| "frac_reward_zero_std": 0.484375, |
| "grad_norm": 0.08832062141924793, |
| "learning_rate": 2.4000000000000003e-06, |
| "loss": 0.0098, |
| "num_tokens": 4957852.0, |
| "reward": 0.3430989682674408, |
| "reward_std": 0.4748988151550293, |
| "rewards/reward_correctness/mean": 0.3430989682674408, |
| "rewards/reward_correctness/std": 0.4748988151550293, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999831318855286, |
| "sampling/importance_sampling_ratio/min": 0.10895482450723648, |
| "sampling/sampling_logp_difference/max": 2.2168219089508057, |
| "sampling/sampling_logp_difference/mean": 0.014919000677764416, |
| "step": 5, |
| "step_time": 286.02837926428765 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.004557291977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2260.0, |
| "completions/mean_length": 525.5950927734375, |
| "completions/mean_terminated_length": 513.9371948242188, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.24871817359235138, |
| "epoch": 0.08823529411764706, |
| "frac_reward_zero_std": 0.4765625, |
| "grad_norm": 0.09720010779023536, |
| "learning_rate": 3e-06, |
| "loss": 0.0129, |
| "num_tokens": 5911986.0, |
| "reward": 0.388671875, |
| "reward_std": 0.4876072406768799, |
| "rewards/reward_correctness/mean": 0.388671875, |
| "rewards/reward_correctness/std": 0.48760727047920227, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000280141830444, |
| "sampling/importance_sampling_ratio/min": 0.06059809401631355, |
| "sampling/sampling_logp_difference/max": 2.8034918308258057, |
| "sampling/sampling_logp_difference/mean": 0.015820473432540894, |
| "step": 6, |
| "step_time": 278.1581368963234 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0078125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2586.0, |
| "completions/mean_length": 557.4264526367188, |
| "completions/mean_terminated_length": 537.6266479492188, |
| "completions/min_length": 83.0, |
| "completions/min_terminated_length": 83.0, |
| "entropy": 0.24867978470865637, |
| "epoch": 0.10294117647058823, |
| "frac_reward_zero_std": 0.515625, |
| "grad_norm": 0.0920658987432061, |
| "learning_rate": 2.9996118137817615e-06, |
| "loss": 0.0052, |
| "num_tokens": 6918073.0, |
| "reward": 0.39453125, |
| "reward_std": 0.48890894651412964, |
| "rewards/reward_correctness/mean": 0.39453125, |
| "rewards/reward_correctness/std": 0.48890894651412964, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000068187713623, |
| "sampling/importance_sampling_ratio/min": 0.055469363927841187, |
| "sampling/sampling_logp_difference/max": 2.8919243812561035, |
| "sampling/sampling_logp_difference/mean": 0.015951288864016533, |
| "step": 7, |
| "step_time": 283.01500198384747 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3047.0, |
| "completions/mean_length": 644.6015625, |
| "completions/mean_terminated_length": 606.0714111328125, |
| "completions/min_length": 152.0, |
| "completions/min_terminated_length": 152.0, |
| "entropy": 0.2523061840329319, |
| "epoch": 0.11764705882352941, |
| "frac_reward_zero_std": 0.4765625, |
| "grad_norm": 0.08226176001367254, |
| "learning_rate": 2.998447478369329e-06, |
| "loss": 0.0112, |
| "num_tokens": 8052013.0, |
| "reward": 0.3118489682674408, |
| "reward_std": 0.46339935064315796, |
| "rewards/reward_correctness/mean": 0.3118489682674408, |
| "rewards/reward_correctness/std": 0.46339938044548035, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000117301940918, |
| "sampling/importance_sampling_ratio/min": 0.11431955546140671, |
| "sampling/sampling_logp_difference/max": 2.168757677078247, |
| "sampling/sampling_logp_difference/mean": 0.015464004129171371, |
| "step": 8, |
| "step_time": 282.23121579224244 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01302083395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2863.0, |
| "completions/mean_length": 615.9193115234375, |
| "completions/mean_terminated_length": 583.5171508789062, |
| "completions/min_length": 157.0, |
| "completions/min_terminated_length": 157.0, |
| "entropy": 0.24067589035257697, |
| "epoch": 0.1323529411764706, |
| "frac_reward_zero_std": 0.4375, |
| "grad_norm": 0.09128268806596039, |
| "learning_rate": 2.9965076633611604e-06, |
| "loss": 0.0138, |
| "num_tokens": 9143985.0, |
| "reward": 0.35546875, |
| "reward_std": 0.47881096601486206, |
| "rewards/reward_correctness/mean": 0.35546875, |
| "rewards/reward_correctness/std": 0.47881099581718445, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000066757202148, |
| "sampling/importance_sampling_ratio/min": 0.20224328339099884, |
| "sampling/sampling_logp_difference/max": 1.5982838869094849, |
| "sampling/sampling_logp_difference/mean": 0.014961779117584229, |
| "step": 9, |
| "step_time": 293.5717437211424 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01692708395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3064.0, |
| "completions/mean_length": 622.6705932617188, |
| "completions/mean_terminated_length": 580.5986938476562, |
| "completions/min_length": 142.0, |
| "completions/min_terminated_length": 142.0, |
| "entropy": 0.22710344230290502, |
| "epoch": 0.14705882352941177, |
| "frac_reward_zero_std": 0.4296875, |
| "grad_norm": 0.09245324413811332, |
| "learning_rate": 2.993793484326816e-06, |
| "loss": 0.0046, |
| "num_tokens": 10266823.0, |
| "reward": 0.3509114682674408, |
| "reward_std": 0.4774106740951538, |
| "rewards/reward_correctness/mean": 0.3509114682674408, |
| "rewards/reward_correctness/std": 0.4774107038974762, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000184774398804, |
| "sampling/importance_sampling_ratio/min": 0.18761380016803741, |
| "sampling/sampling_logp_difference/max": 1.6733696460723877, |
| "sampling/sampling_logp_difference/mean": 0.014758117496967316, |
| "step": 10, |
| "step_time": 302.41576358349994 |
| }, |
| { |
| "epoch": 0.14705882352941177, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.011904762259551458, |
| "eval_completions/max_length": 1140.107142857143, |
| "eval_completions/max_terminated_length": 993.6309523809524, |
| "eval_completions/mean_length": 548.2758098783947, |
| "eval_completions/mean_terminated_length": 518.2567600068592, |
| "eval_completions/min_length": 244.98809523809524, |
| "eval_completions/min_terminated_length": 244.98809523809524, |
| "eval_entropy": 0.19089051132046042, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 10266823.0, |
| "eval_reward": 0.4801587403884956, |
| "eval_reward_std": 0.48688212249960217, |
| "eval_rewards/reward_correctness/mean": 0.48015873471186277, |
| "eval_rewards/reward_correctness/std": 0.48688211966128575, |
| "eval_runtime": 674.283, |
| "eval_samples_per_second": 0.742, |
| "eval_sampling/importance_sampling_ratio/max": 3.0, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9907156576712927, |
| "eval_sampling/importance_sampling_ratio/min": 0.4924284939964612, |
| "eval_sampling/sampling_logp_difference/max": 2.736755425021762, |
| "eval_sampling/sampling_logp_difference/mean": 0.048229864931532314, |
| "eval_steps_per_second": 0.125, |
| "step": 10 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01888020895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3035.0, |
| "completions/mean_length": 645.2077026367188, |
| "completions/mean_terminated_length": 598.5076293945312, |
| "completions/min_length": 98.0, |
| "completions/min_terminated_length": 98.0, |
| "entropy": 0.24283059418667108, |
| "epoch": 0.16176470588235295, |
| "frac_reward_zero_std": 0.421875, |
| "grad_norm": 0.0974447715454529, |
| "learning_rate": 2.990306502165398e-06, |
| "loss": 0.0102, |
| "num_tokens": 11415890.0, |
| "reward": 0.3795573115348816, |
| "reward_std": 0.48543480038642883, |
| "rewards/reward_correctness/mean": 0.3795572817325592, |
| "rewards/reward_correctness/std": 0.4854348301887512, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000299215316772, |
| "sampling/importance_sampling_ratio/min": 0.14718766510486603, |
| "sampling/sampling_logp_difference/max": 1.9160468578338623, |
| "sampling/sampling_logp_difference/mean": 0.015639059245586395, |
| "step": 11, |
| "step_time": 291.7094019954093 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0201822929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2988.0, |
| "completions/mean_length": 683.3971557617188, |
| "completions/mean_terminated_length": 634.1966552734375, |
| "completions/min_length": 140.0, |
| "completions/min_terminated_length": 140.0, |
| "entropy": 0.25403576996177435, |
| "epoch": 0.17647058823529413, |
| "frac_reward_zero_std": 0.4453125, |
| "grad_norm": 0.09531950301623979, |
| "learning_rate": 2.986048722207899e-06, |
| "loss": 0.012, |
| "num_tokens": 12632868.0, |
| "reward": 0.380859375, |
| "reward_std": 0.4857562184333801, |
| "rewards/reward_correctness/mean": 0.380859375, |
| "rewards/reward_correctness/std": 0.4857562482357025, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999941885471344, |
| "sampling/importance_sampling_ratio/min": 0.05190279707312584, |
| "sampling/sampling_logp_difference/max": 2.9583826065063477, |
| "sampling/sampling_logp_difference/mean": 0.01636284403502941, |
| "step": 12, |
| "step_time": 290.1786908484064 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0162760429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3024.0, |
| "completions/mean_length": 671.1953125, |
| "completions/mean_terminated_length": 633.8160400390625, |
| "completions/min_length": 128.0, |
| "completions/min_terminated_length": 128.0, |
| "entropy": 0.23784961050841957, |
| "epoch": 0.19117647058823528, |
| "frac_reward_zero_std": 0.40625, |
| "grad_norm": 0.09189129853735502, |
| "learning_rate": 2.981022593063946e-06, |
| "loss": 0.0089, |
| "num_tokens": 13847616.0, |
| "reward": 0.357421875, |
| "reward_std": 0.4793965518474579, |
| "rewards/reward_correctness/mean": 0.357421875, |
| "rewards/reward_correctness/std": 0.47939661145210266, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9972469210624695, |
| "sampling/importance_sampling_ratio/min": 1.9157713474415514e-14, |
| "sampling/sampling_logp_difference/max": 31.586071014404297, |
| "sampling/sampling_logp_difference/mean": 0.04794478043913841, |
| "step": 13, |
| "step_time": 341.73873996781185 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02083333395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3065.0, |
| "completions/mean_length": 707.0162963867188, |
| "completions/mean_terminated_length": 656.6974487304688, |
| "completions/min_length": 126.0, |
| "completions/min_terminated_length": 126.0, |
| "entropy": 0.2236228445544839, |
| "epoch": 0.20588235294117646, |
| "frac_reward_zero_std": 0.46875, |
| "grad_norm": 0.08294063821997087, |
| "learning_rate": 2.9752310052136353e-06, |
| "loss": 0.0156, |
| "num_tokens": 15082525.0, |
| "reward": 0.3951823115348816, |
| "reward_std": 0.4890490174293518, |
| "rewards/reward_correctness/mean": 0.3951822817325592, |
| "rewards/reward_correctness/std": 0.4890490770339966, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000492334365845, |
| "sampling/importance_sampling_ratio/min": 0.06042155250906944, |
| "sampling/sampling_logp_difference/max": 2.8064093589782715, |
| "sampling/sampling_logp_difference/mean": 0.015129472129046917, |
| "step": 14, |
| "step_time": 307.8643926582299 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2959.0, |
| "completions/mean_length": 675.498046875, |
| "completions/mean_terminated_length": 637.7943115234375, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.23216103401500732, |
| "epoch": 0.22058823529411764, |
| "frac_reward_zero_std": 0.453125, |
| "grad_norm": 0.10101422201216483, |
| "learning_rate": 2.968677289345242e-06, |
| "loss": 0.0107, |
| "num_tokens": 16288186.0, |
| "reward": 0.3997395932674408, |
| "reward_std": 0.49000421166419983, |
| "rewards/reward_correctness/mean": 0.3997395932674408, |
| "rewards/reward_correctness/std": 0.4900042414665222, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9976019859313965, |
| "sampling/importance_sampling_ratio/min": 5.457596647330436e-18, |
| "sampling/sampling_logp_difference/max": 39.7495231628418, |
| "sampling/sampling_logp_difference/mean": 0.04638348147273064, |
| "step": 15, |
| "step_time": 320.05006989324465 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02734375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3016.0, |
| "completions/mean_length": 722.0931396484375, |
| "completions/mean_terminated_length": 656.0314331054688, |
| "completions/min_length": 115.0, |
| "completions/min_terminated_length": 115.0, |
| "entropy": 0.24111876578535885, |
| "epoch": 0.23529411764705882, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.08050920134854482, |
| "learning_rate": 2.9613652144397706e-06, |
| "loss": 0.0131, |
| "num_tokens": 17574597.0, |
| "reward": 0.361328125, |
| "reward_std": 0.48054182529449463, |
| "rewards/reward_correctness/mean": 0.361328125, |
| "rewards/reward_correctness/std": 0.48054182529449463, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000146627426147, |
| "sampling/importance_sampling_ratio/min": 0.08782368153333664, |
| "sampling/sampling_logp_difference/max": 2.4324240684509277, |
| "sampling/sampling_logp_difference/mean": 0.015971485525369644, |
| "step": 16, |
| "step_time": 324.0193057116121 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.017578125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3001.0, |
| "completions/mean_length": 711.9896240234375, |
| "completions/mean_terminated_length": 669.7627563476562, |
| "completions/min_length": 131.0, |
| "completions/min_terminated_length": 131.0, |
| "entropy": 0.22629119083285332, |
| "epoch": 0.25, |
| "frac_reward_zero_std": 0.4609375, |
| "grad_norm": 0.0900011417432106, |
| "learning_rate": 2.9532989856034515e-06, |
| "loss": 0.0086, |
| "num_tokens": 18819317.0, |
| "reward": 0.38671875, |
| "reward_std": 0.4871568977832794, |
| "rewards/reward_correctness/mean": 0.38671875, |
| "rewards/reward_correctness/std": 0.4871569275856018, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.997471809387207, |
| "sampling/importance_sampling_ratio/min": 3.498903030914151e-14, |
| "sampling/sampling_logp_difference/max": 30.983741760253906, |
| "sampling/sampling_logp_difference/mean": 0.045513179153203964, |
| "step": 17, |
| "step_time": 328.7708681197837 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.025390625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3011.0, |
| "completions/mean_length": 725.828125, |
| "completions/mean_terminated_length": 664.7054443359375, |
| "completions/min_length": 138.0, |
| "completions/min_terminated_length": 138.0, |
| "entropy": 0.22558519581798464, |
| "epoch": 0.2647058823529412, |
| "frac_reward_zero_std": 0.421875, |
| "grad_norm": 0.09060650894291081, |
| "learning_rate": 2.94448324164942e-06, |
| "loss": 0.01, |
| "num_tokens": 20101097.0, |
| "reward": 0.3626302182674408, |
| "reward_std": 0.4809158742427826, |
| "rewards/reward_correctness/mean": 0.3626302182674408, |
| "rewards/reward_correctness/std": 0.480915904045105, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000027418136597, |
| "sampling/importance_sampling_ratio/min": 0.1281590312719345, |
| "sampling/sampling_logp_difference/max": 2.4219703674316406, |
| "sampling/sampling_logp_difference/mean": 0.015497544780373573, |
| "step": 18, |
| "step_time": 311.72907185554504 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0221354179084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3016.0, |
| "completions/mean_length": 702.263671875, |
| "completions/mean_terminated_length": 648.6211547851562, |
| "completions/min_length": 148.0, |
| "completions/min_terminated_length": 148.0, |
| "entropy": 0.2129860440036282, |
| "epoch": 0.27941176470588236, |
| "frac_reward_zero_std": 0.4296875, |
| "grad_norm": 0.0864612338873769, |
| "learning_rate": 2.934923052429984e-06, |
| "loss": 0.0126, |
| "num_tokens": 21336422.0, |
| "reward": 0.416015625, |
| "reward_std": 0.49305665493011475, |
| "rewards/reward_correctness/mean": 0.416015625, |
| "rewards/reward_correctness/std": 0.49305668473243713, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000278949737549, |
| "sampling/importance_sampling_ratio/min": 0.19384071230888367, |
| "sampling/sampling_logp_difference/max": 2.108778953552246, |
| "sampling/sampling_logp_difference/mean": 0.01505441963672638, |
| "step": 19, |
| "step_time": 298.8261763001792 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0221354179084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3009.0, |
| "completions/mean_length": 698.8724365234375, |
| "completions/mean_terminated_length": 645.153076171875, |
| "completions/min_length": 132.0, |
| "completions/min_terminated_length": 132.0, |
| "entropy": 0.21506994613446295, |
| "epoch": 0.29411764705882354, |
| "frac_reward_zero_std": 0.4296875, |
| "grad_norm": 0.09299737935420878, |
| "learning_rate": 2.924623915920992e-06, |
| "loss": 0.0199, |
| "num_tokens": 22553890.0, |
| "reward": 0.3893229365348816, |
| "reward_std": 0.48775550723075867, |
| "rewards/reward_correctness/mean": 0.3893229067325592, |
| "rewards/reward_correctness/std": 0.48775556683540344, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000158548355103, |
| "sampling/importance_sampling_ratio/min": 0.051003288477659225, |
| "sampling/sampling_logp_difference/max": 2.975865125656128, |
| "sampling/sampling_logp_difference/mean": 0.01541055552661419, |
| "step": 20, |
| "step_time": 296.1154873804189 |
| }, |
| { |
| "epoch": 0.29411764705882354, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.00992063521629288, |
| "eval_completions/max_length": 1160.6309523809523, |
| "eval_completions/max_terminated_length": 1036.6785714285713, |
| "eval_completions/mean_length": 563.2639083862305, |
| "eval_completions/mean_terminated_length": 538.8448602585565, |
| "eval_completions/min_length": 253.11904761904762, |
| "eval_completions/min_terminated_length": 253.11904761904762, |
| "eval_entropy": 0.16844143558825767, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 22553890.0, |
| "eval_reward": 0.5218254091838995, |
| "eval_reward_std": 0.47886995829287027, |
| "eval_rewards/reward_correctness/mean": 0.5218254020881086, |
| "eval_rewards/reward_correctness/std": 0.47886995545455385, |
| "eval_runtime": 684.1716, |
| "eval_samples_per_second": 0.731, |
| "eval_sampling/importance_sampling_ratio/max": 3.0, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9925182844911303, |
| "eval_sampling/importance_sampling_ratio/min": 0.5034990778991154, |
| "eval_sampling/sampling_logp_difference/max": 2.5275759952408925, |
| "eval_sampling/sampling_logp_difference/mean": 0.04347030202015525, |
| "eval_steps_per_second": 0.123, |
| "step": 20 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0319010429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3067.0, |
| "completions/mean_length": 783.7174682617188, |
| "completions/mean_terminated_length": 708.4525756835938, |
| "completions/min_length": 169.0, |
| "completions/min_terminated_length": 169.0, |
| "entropy": 0.21329337812494487, |
| "epoch": 0.3088235294117647, |
| "frac_reward_zero_std": 0.4453125, |
| "grad_norm": 0.08555682284566928, |
| "learning_rate": 2.913591755060004e-06, |
| "loss": 0.0163, |
| "num_tokens": 23924636.0, |
| "reward": 0.3626302182674408, |
| "reward_std": 0.4809158742427826, |
| "rewards/reward_correctness/mean": 0.3626302182674408, |
| "rewards/reward_correctness/std": 0.480915904045105, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000237226486206, |
| "sampling/importance_sampling_ratio/min": 0.17430639266967773, |
| "sampling/sampling_logp_difference/max": 1.7469406127929688, |
| "sampling/sampling_logp_difference/mean": 0.015152385458350182, |
| "step": 21, |
| "step_time": 323.7390418453142 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0345052108168602, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3058.0, |
| "completions/mean_length": 776.927734375, |
| "completions/mean_terminated_length": 694.9056396484375, |
| "completions/min_length": 182.0, |
| "completions/min_terminated_length": 182.0, |
| "entropy": 0.20884651492815465, |
| "epoch": 0.3235294117647059, |
| "frac_reward_zero_std": 0.4453125, |
| "grad_norm": 0.08504344406668593, |
| "learning_rate": 2.901832914340062e-06, |
| "loss": 0.0145, |
| "num_tokens": 25266329.0, |
| "reward": 0.3463541865348816, |
| "reward_std": 0.4759626090526581, |
| "rewards/reward_correctness/mean": 0.3463541567325592, |
| "rewards/reward_correctness/std": 0.47596266865730286, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999927818775177, |
| "sampling/importance_sampling_ratio/min": 0.10210206359624863, |
| "sampling/sampling_logp_difference/max": 2.281782388687134, |
| "sampling/sampling_logp_difference/mean": 0.015026840381324291, |
| "step": 22, |
| "step_time": 308.0862176595256 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0221354179084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3059.0, |
| "completions/mean_length": 682.2506713867188, |
| "completions/mean_terminated_length": 628.1550903320312, |
| "completions/min_length": 150.0, |
| "completions/min_terminated_length": 150.0, |
| "entropy": 0.18090229877270758, |
| "epoch": 0.3382352941176471, |
| "frac_reward_zero_std": 0.46875, |
| "grad_norm": 0.08417650985796771, |
| "learning_rate": 2.889354156161033e-06, |
| "loss": 0.0037, |
| "num_tokens": 26476398.0, |
| "reward": 0.40234375, |
| "reward_std": 0.49053022265434265, |
| "rewards/reward_correctness/mean": 0.40234375, |
| "rewards/reward_correctness/std": 0.49053022265434265, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9974940419197083, |
| "sampling/importance_sampling_ratio/min": 2.3660272959608042e-14, |
| "sampling/sampling_logp_difference/max": 31.37497901916504, |
| "sampling/sampling_logp_difference/mean": 0.0386495515704155, |
| "step": 23, |
| "step_time": 333.6117503368296 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02083333395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2987.0, |
| "completions/mean_length": 681.4134521484375, |
| "completions/mean_terminated_length": 630.5498657226562, |
| "completions/min_length": 158.0, |
| "completions/min_terminated_length": 158.0, |
| "entropy": 0.1827188755851239, |
| "epoch": 0.35294117647058826, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.0813981211251326, |
| "learning_rate": 2.876162656940614e-06, |
| "loss": 0.0063, |
| "num_tokens": 27691781.0, |
| "reward": 0.4134114682674408, |
| "reward_std": 0.49260568618774414, |
| "rewards/reward_correctness/mean": 0.4134114682674408, |
| "rewards/reward_correctness/std": 0.49260571599006653, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999978244304657, |
| "sampling/importance_sampling_ratio/min": 0.033229220658540726, |
| "sampling/sampling_logp_difference/max": 3.4043257236480713, |
| "sampling/sampling_logp_difference/mean": 0.013785876333713531, |
| "step": 24, |
| "step_time": 318.87752299709246 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0162760429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2978.0, |
| "completions/mean_length": 664.8314208984375, |
| "completions/mean_terminated_length": 625.0039672851562, |
| "completions/min_length": 136.0, |
| "completions/min_terminated_length": 136.0, |
| "entropy": 0.17039690352976322, |
| "epoch": 0.36764705882352944, |
| "frac_reward_zero_std": 0.4609375, |
| "grad_norm": 0.08972646850895903, |
| "learning_rate": 2.862266002987244e-06, |
| "loss": 0.0086, |
| "num_tokens": 28863442.0, |
| "reward": 0.4075520932674408, |
| "reward_std": 0.4915390610694885, |
| "rewards/reward_correctness/mean": 0.4075520932674408, |
| "rewards/reward_correctness/std": 0.4915390610694885, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999510645866394, |
| "sampling/importance_sampling_ratio/min": 0.050326716154813766, |
| "sampling/sampling_logp_difference/max": 2.9892191886901855, |
| "sampling/sampling_logp_difference/mean": 0.013295136392116547, |
| "step": 25, |
| "step_time": 308.65431338362396 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01953125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3001.0, |
| "completions/mean_length": 694.8568115234375, |
| "completions/mean_terminated_length": 647.5033569335938, |
| "completions/min_length": 129.0, |
| "completions/min_terminated_length": 129.0, |
| "entropy": 0.1789090399397537, |
| "epoch": 0.38235294117647056, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.09002986293169273, |
| "learning_rate": 2.847672186137282e-06, |
| "loss": 0.014, |
| "num_tokens": 30105822.0, |
| "reward": 0.359375, |
| "reward_std": 0.47997352480888367, |
| "rewards/reward_correctness/mean": 0.359375, |
| "rewards/reward_correctness/std": 0.47997352480888367, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999814033508301, |
| "sampling/importance_sampling_ratio/min": 0.10588235408067703, |
| "sampling/sampling_logp_difference/max": 2.4165873527526855, |
| "sampling/sampling_logp_difference/mean": 0.013810048811137676, |
| "step": 26, |
| "step_time": 317.5119425994344 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02669270895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2983.0, |
| "completions/mean_length": 712.9296875, |
| "completions/mean_terminated_length": 648.2327880859375, |
| "completions/min_length": 114.0, |
| "completions/min_terminated_length": 114.0, |
| "entropy": 0.17184418614488095, |
| "epoch": 0.39705882352941174, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.09041602003319889, |
| "learning_rate": 2.8323895991589866e-06, |
| "loss": 0.0023, |
| "num_tokens": 31355742.0, |
| "reward": 0.4388020932674408, |
| "reward_std": 0.4964022934436798, |
| "rewards/reward_correctness/mean": 0.4388020932674408, |
| "rewards/reward_correctness/std": 0.4964022934436798, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000118017196655, |
| "sampling/importance_sampling_ratio/min": 0.03874439746141434, |
| "sampling/sampling_logp_difference/max": 3.2507691383361816, |
| "sampling/sampling_logp_difference/mean": 0.013423329219222069, |
| "step": 27, |
| "step_time": 312.58686931058764 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01692708395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2944.0, |
| "completions/mean_length": 640.52734375, |
| "completions/mean_terminated_length": 598.6609497070312, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.1618180574150756, |
| "epoch": 0.4117647058823529, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.09128044036460715, |
| "learning_rate": 2.8164270309259034e-06, |
| "loss": 0.0073, |
| "num_tokens": 32496000.0, |
| "reward": 0.44921875, |
| "reward_std": 0.4975765645503998, |
| "rewards/reward_correctness/mean": 0.44921875, |
| "rewards/reward_correctness/std": 0.4975765645503998, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9966862201690674, |
| "sampling/importance_sampling_ratio/min": 2.0281311352565723e-12, |
| "sampling/sampling_logp_difference/max": 26.923906326293945, |
| "sampling/sampling_logp_difference/mean": 0.04643641412258148, |
| "step": 28, |
| "step_time": 353.90896678948775 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.012369791977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2864.0, |
| "completions/mean_length": 649.8665771484375, |
| "completions/mean_terminated_length": 619.5299682617188, |
| "completions/min_length": 142.0, |
| "completions/min_terminated_length": 142.0, |
| "entropy": 0.17093808809295297, |
| "epoch": 0.4264705882352941, |
| "frac_reward_zero_std": 0.4921875, |
| "grad_norm": 0.08931129502738985, |
| "learning_rate": 2.79979366136247e-06, |
| "loss": 0.0086, |
| "num_tokens": 33646175.0, |
| "reward": 0.37890625, |
| "reward_std": 0.48527270555496216, |
| "rewards/reward_correctness/mean": 0.37890625, |
| "rewards/reward_correctness/std": 0.48527273535728455, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999492764472961, |
| "sampling/importance_sampling_ratio/min": 0.07345923781394958, |
| "sampling/sampling_logp_difference/max": 2.6110246181488037, |
| "sampling/sampling_logp_difference/mean": 0.013474516570568085, |
| "step": 29, |
| "step_time": 295.66233261767775 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01953125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3038.0, |
| "completions/mean_length": 661.0436401367188, |
| "completions/mean_terminated_length": 613.0166015625, |
| "completions/min_length": 137.0, |
| "completions/min_terminated_length": 137.0, |
| "entropy": 0.16064971894957125, |
| "epoch": 0.4411764705882353, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.09301717877462062, |
| "learning_rate": 2.7824990561647276e-06, |
| "loss": 0.0064, |
| "num_tokens": 34807266.0, |
| "reward": 0.4016927182674408, |
| "reward_std": 0.4904000461101532, |
| "rewards/reward_correctness/mean": 0.4016927182674408, |
| "rewards/reward_correctness/std": 0.4904000759124756, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000295639038086, |
| "sampling/importance_sampling_ratio/min": 4.900506974081509e-05, |
| "sampling/sampling_logp_difference/max": 9.92358684539795, |
| "sampling/sampling_logp_difference/mean": 0.0127759650349617, |
| "step": 30, |
| "step_time": 331.1905814567581 |
| }, |
| { |
| "epoch": 0.4411764705882353, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.023809524519102915, |
| "eval_completions/max_length": 1353.3809523809523, |
| "eval_completions/max_terminated_length": 1045.2619047619048, |
| "eval_completions/mean_length": 598.2460470653716, |
| "eval_completions/mean_terminated_length": 538.2575521923247, |
| "eval_completions/min_length": 258.3095238095238, |
| "eval_completions/min_terminated_length": 258.3095238095238, |
| "eval_entropy": 0.1339410948788836, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 34807266.0, |
| "eval_reward": 0.5178571552747772, |
| "eval_reward_std": 0.49191097028198694, |
| "eval_rewards/reward_correctness/mean": 0.517857147469407, |
| "eval_rewards/reward_correctness/std": 0.4919109692176183, |
| "eval_runtime": 806.339, |
| "eval_samples_per_second": 0.62, |
| "eval_sampling/importance_sampling_ratio/max": 3.0, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9940516991274697, |
| "eval_sampling/importance_sampling_ratio/min": 0.5017738665143648, |
| "eval_sampling/sampling_logp_difference/max": 2.459620109626225, |
| "eval_sampling/sampling_logp_difference/mean": 0.0351627429370724, |
| "eval_steps_per_second": 0.104, |
| "step": 30 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01888020895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2916.0, |
| "completions/mean_length": 652.9173583984375, |
| "completions/mean_terminated_length": 606.3656616210938, |
| "completions/min_length": 140.0, |
| "completions/min_terminated_length": 140.0, |
| "entropy": 0.1553694016765803, |
| "epoch": 0.45588235294117646, |
| "frac_reward_zero_std": 0.4921875, |
| "grad_norm": 0.0903234273155232, |
| "learning_rate": 2.7645531612991763e-06, |
| "loss": 0.0091, |
| "num_tokens": 35967767.0, |
| "reward": 0.435546875, |
| "reward_std": 0.4959898591041565, |
| "rewards/reward_correctness/mean": 0.435546875, |
| "rewards/reward_correctness/std": 0.4959898591041565, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999983012676239, |
| "sampling/importance_sampling_ratio/min": 0.036659516394138336, |
| "sampling/sampling_logp_difference/max": 3.306082248687744, |
| "sampling/sampling_logp_difference/mean": 0.01269611157476902, |
| "step": 31, |
| "step_time": 314.0296438266523 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.008463541977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3054.0, |
| "completions/mean_length": 680.3131713867188, |
| "completions/mean_terminated_length": 659.8982543945312, |
| "completions/min_length": 124.0, |
| "completions/min_terminated_length": 124.0, |
| "entropy": 0.1594469680567272, |
| "epoch": 0.47058823529411764, |
| "frac_reward_zero_std": 0.5390625, |
| "grad_norm": 0.08553326807932113, |
| "learning_rate": 2.745966297282944e-06, |
| "loss": 0.0035, |
| "num_tokens": 37146348.0, |
| "reward": 0.3802083432674408, |
| "reward_std": 0.4855959713459015, |
| "rewards/reward_correctness/mean": 0.3802083432674408, |
| "rewards/reward_correctness/std": 0.4855960011482239, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000026822090149, |
| "sampling/importance_sampling_ratio/min": 0.004895252175629139, |
| "sampling/sampling_logp_difference/max": 5.319489479064941, |
| "sampling/sampling_logp_difference/mean": 0.012724189087748528, |
| "step": 32, |
| "step_time": 284.7576134908013 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02083333395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3037.0, |
| "completions/mean_length": 649.078125, |
| "completions/mean_terminated_length": 597.564453125, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.14795389265054837, |
| "epoch": 0.4852941176470588, |
| "frac_reward_zero_std": 0.4921875, |
| "grad_norm": 0.09217934339873941, |
| "learning_rate": 2.726749153248549e-06, |
| "loss": 0.008, |
| "num_tokens": 38312112.0, |
| "reward": 0.439453125, |
| "reward_std": 0.49648213386535645, |
| "rewards/reward_correctness/mean": 0.439453125, |
| "rewards/reward_correctness/std": 0.4964821934700012, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999476075172424, |
| "sampling/importance_sampling_ratio/min": 0.09122932702302933, |
| "sampling/sampling_logp_difference/max": 2.394378900527954, |
| "sampling/sampling_logp_difference/mean": 0.012328311800956726, |
| "step": 33, |
| "step_time": 312.3779431232251 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.012369791977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2798.0, |
| "completions/mean_length": 613.1237182617188, |
| "completions/mean_terminated_length": 582.3269653320312, |
| "completions/min_length": 146.0, |
| "completions/min_terminated_length": 146.0, |
| "entropy": 0.1546164610190317, |
| "epoch": 0.5, |
| "frac_reward_zero_std": 0.4609375, |
| "grad_norm": 0.10097534688518955, |
| "learning_rate": 2.706912780796687e-06, |
| "loss": 0.0144, |
| "num_tokens": 39437566.0, |
| "reward": 0.46875, |
| "reward_std": 0.4991849958896637, |
| "rewards/reward_correctness/mean": 0.46875, |
| "rewards/reward_correctness/std": 0.4991849958896637, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000132322311401, |
| "sampling/importance_sampling_ratio/min": 0.0012084051268175244, |
| "sampling/sampling_logp_difference/max": 6.718453884124756, |
| "sampling/sampling_logp_difference/mean": 0.012729411944746971, |
| "step": 34, |
| "step_time": 299.6743005514145 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.009765625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2971.0, |
| "completions/mean_length": 645.4049682617188, |
| "completions/mean_terminated_length": 621.4740600585938, |
| "completions/min_length": 161.0, |
| "completions/min_terminated_length": 161.0, |
| "entropy": 0.15254527825163677, |
| "epoch": 0.5147058823529411, |
| "frac_reward_zero_std": 0.390625, |
| "grad_norm": 0.09941734222697605, |
| "learning_rate": 2.686468587640551e-06, |
| "loss": 0.0153, |
| "num_tokens": 40596032.0, |
| "reward": 0.4173177182674408, |
| "reward_std": 0.49327680468559265, |
| "rewards/reward_correctness/mean": 0.4173177182674408, |
| "rewards/reward_correctness/std": 0.49327683448791504, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000033378601074, |
| "sampling/importance_sampling_ratio/min": 0.06955904513597488, |
| "sampling/sampling_logp_difference/max": 2.665579319000244, |
| "sampling/sampling_logp_difference/mean": 0.012409794144332409, |
| "step": 35, |
| "step_time": 309.1410033633001 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.03059895895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2755.0, |
| "completions/mean_length": 697.9694213867188, |
| "completions/mean_terminated_length": 623.3291015625, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.16129665705375373, |
| "epoch": 0.5294117647058824, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.08853224454748279, |
| "learning_rate": 2.6654283310453644e-06, |
| "loss": 0.0129, |
| "num_tokens": 41840481.0, |
| "reward": 0.41015625, |
| "reward_std": 0.4920220673084259, |
| "rewards/reward_correctness/mean": 0.41015625, |
| "rewards/reward_correctness/std": 0.4920220673084259, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000028133392334, |
| "sampling/importance_sampling_ratio/min": 0.054633643478155136, |
| "sampling/sampling_logp_difference/max": 2.9071054458618164, |
| "sampling/sampling_logp_difference/mean": 0.012762945145368576, |
| "step": 36, |
| "step_time": 331.5215606596321 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.014322916977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2977.0, |
| "completions/mean_length": 650.126953125, |
| "completions/mean_terminated_length": 615.03369140625, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.13540291565004736, |
| "epoch": 0.5441176470588235, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.09558218428111698, |
| "learning_rate": 2.643804111066888e-06, |
| "loss": 0.0004, |
| "num_tokens": 42998640.0, |
| "reward": 0.4524739682674408, |
| "reward_std": 0.49789825081825256, |
| "rewards/reward_correctness/mean": 0.4524739682674408, |
| "rewards/reward_correctness/std": 0.49789825081825256, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9963769316673279, |
| "sampling/importance_sampling_ratio/min": 2.2342668740924247e-15, |
| "sampling/sampling_logp_difference/max": 33.73486328125, |
| "sampling/sampling_logp_difference/mean": 0.04356236383318901, |
| "step": 37, |
| "step_time": 318.7183957109228 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.013671875, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3070.0, |
| "completions/mean_length": 639.828125, |
| "completions/mean_terminated_length": 606.1597290039062, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.14192722243024036, |
| "epoch": 0.5588235294117647, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.09128377744038438, |
| "learning_rate": 2.6216083635927896e-06, |
| "loss": 0.0145, |
| "num_tokens": 44127948.0, |
| "reward": 0.4055989682674408, |
| "reward_std": 0.49116745591163635, |
| "rewards/reward_correctness/mean": 0.4055989682674408, |
| "rewards/reward_correctness/std": 0.49116748571395874, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999954104423523, |
| "sampling/importance_sampling_ratio/min": 0.12891176342964172, |
| "sampling/sampling_logp_difference/max": 2.07913875579834, |
| "sampling/sampling_logp_difference/mean": 0.0118638901039958, |
| "step": 38, |
| "step_time": 291.72405133536085 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01106770895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3021.0, |
| "completions/mean_length": 651.6712646484375, |
| "completions/mean_terminated_length": 624.5839233398438, |
| "completions/min_length": 122.0, |
| "completions/min_terminated_length": 122.0, |
| "entropy": 0.144097798562143, |
| "epoch": 0.5735294117647058, |
| "frac_reward_zero_std": 0.453125, |
| "grad_norm": 0.09866853100849818, |
| "learning_rate": 2.598853853190882e-06, |
| "loss": 0.0117, |
| "num_tokens": 45284507.0, |
| "reward": 0.4303385615348816, |
| "reward_std": 0.4952847361564636, |
| "rewards/reward_correctness/mean": 0.4303385317325592, |
| "rewards/reward_correctness/std": 0.495284765958786, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000274181365967, |
| "sampling/importance_sampling_ratio/min": 0.07444321364164352, |
| "sampling/sampling_logp_difference/max": 2.5977187156677246, |
| "sampling/sampling_logp_difference/mean": 0.012085234746336937, |
| "step": 39, |
| "step_time": 307.95729614887387 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.010416666977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2808.0, |
| "completions/mean_length": 647.4193115234375, |
| "completions/mean_terminated_length": 622.0487060546875, |
| "completions/min_length": 126.0, |
| "completions/min_terminated_length": 126.0, |
| "entropy": 0.14093391405185685, |
| "epoch": 0.5882352941176471, |
| "frac_reward_zero_std": 0.484375, |
| "grad_norm": 0.09003847468670717, |
| "learning_rate": 2.5755536657683354e-06, |
| "loss": 0.0038, |
| "num_tokens": 46434355.0, |
| "reward": 0.4361979365348816, |
| "reward_std": 0.49607405066490173, |
| "rewards/reward_correctness/mean": 0.4361979067325592, |
| "rewards/reward_correctness/std": 0.4960741102695465, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999783635139465, |
| "sampling/importance_sampling_ratio/min": 0.0060172248631715775, |
| "sampling/sampling_logp_difference/max": 5.113129138946533, |
| "sampling/sampling_logp_difference/mean": 0.011764663271605968, |
| "step": 40, |
| "step_time": 293.8941194838844 |
| }, |
| { |
| "epoch": 0.5882352941176471, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.007936508173034304, |
| "eval_completions/max_length": 1165.6904761904761, |
| "eval_completions/max_terminated_length": 1080.047619047619, |
| "eval_completions/mean_length": 574.1071610223679, |
| "eval_completions/mean_terminated_length": 554.6643026442755, |
| "eval_completions/min_length": 256.3452380952381, |
| "eval_completions/min_terminated_length": 256.3452380952381, |
| "eval_entropy": 0.1198243172395797, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 46434355.0, |
| "eval_reward": 0.5138889015430496, |
| "eval_reward_std": 0.4920797245133491, |
| "eval_rewards/reward_correctness/mean": 0.5138888944472585, |
| "eval_rewards/reward_correctness/std": 0.4920797199010849, |
| "eval_runtime": 686.0979, |
| "eval_samples_per_second": 0.729, |
| "eval_sampling/importance_sampling_ratio/max": 3.0, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9935774937981651, |
| "eval_sampling/importance_sampling_ratio/min": 0.5091292620414779, |
| "eval_sampling/sampling_logp_difference/max": 2.236756165822347, |
| "eval_sampling/sampling_logp_difference/mean": 0.03119110870396807, |
| "eval_steps_per_second": 0.122, |
| "step": 40 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.014322916977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2812.0, |
| "completions/mean_length": 638.0228271484375, |
| "completions/mean_terminated_length": 602.69287109375, |
| "completions/min_length": 133.0, |
| "completions/min_terminated_length": 133.0, |
| "entropy": 0.14125462470110506, |
| "epoch": 0.6029411764705882, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.0900402655314074, |
| "learning_rate": 2.551721201046098e-06, |
| "loss": 0.0076, |
| "num_tokens": 47570106.0, |
| "reward": 0.421875, |
| "reward_std": 0.49401959776878357, |
| "rewards/reward_correctness/mean": 0.421875, |
| "rewards/reward_correctness/std": 0.49401959776878357, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999483227729797, |
| "sampling/importance_sampling_ratio/min": 0.0491253063082695, |
| "sampling/sampling_logp_difference/max": 3.013381004333496, |
| "sampling/sampling_logp_difference/mean": 0.011780498549342155, |
| "step": 41, |
| "step_time": 294.60718926927075 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01692708395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2950.0, |
| "completions/mean_length": 682.486328125, |
| "completions/mean_terminated_length": 641.3424072265625, |
| "completions/min_length": 162.0, |
| "completions/min_terminated_length": 162.0, |
| "entropy": 0.13389375229598954, |
| "epoch": 0.6176470588235294, |
| "frac_reward_zero_std": 0.515625, |
| "grad_norm": 0.08930218418218254, |
| "learning_rate": 2.5273701648528393e-06, |
| "loss": 0.0104, |
| "num_tokens": 48771609.0, |
| "reward": 0.439453125, |
| "reward_std": 0.49648216366767883, |
| "rewards/reward_correctness/mean": 0.439453125, |
| "rewards/reward_correctness/std": 0.4964821934700012, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999879002571106, |
| "sampling/importance_sampling_ratio/min": 0.04630628600716591, |
| "sampling/sampling_logp_difference/max": 3.0724775791168213, |
| "sampling/sampling_logp_difference/mean": 0.011399103328585625, |
| "step": 42, |
| "step_time": 306.2859174082987 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.012369791977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2893.0, |
| "completions/mean_length": 634.833984375, |
| "completions/mean_terminated_length": 604.3091430664062, |
| "completions/min_length": 127.0, |
| "completions/min_terminated_length": 127.0, |
| "entropy": 0.13599129632348195, |
| "epoch": 0.6323529411764706, |
| "frac_reward_zero_std": 0.546875, |
| "grad_norm": 0.08873768343136257, |
| "learning_rate": 2.5025145612428566e-06, |
| "loss": 0.0022, |
| "num_tokens": 49918734.0, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.4931671619415283, |
| "rewards/reward_correctness/mean": 0.4166666567325592, |
| "rewards/reward_correctness/std": 0.4931672215461731, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999739527702332, |
| "sampling/importance_sampling_ratio/min": 0.04011012613773346, |
| "sampling/sampling_logp_difference/max": 3.2161264419555664, |
| "sampling/sampling_logp_difference/mean": 0.011722831055521965, |
| "step": 43, |
| "step_time": 321.6622120910324 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01888020895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2979.0, |
| "completions/mean_length": 688.9193115234375, |
| "completions/mean_terminated_length": 643.0603637695312, |
| "completions/min_length": 138.0, |
| "completions/min_terminated_length": 138.0, |
| "entropy": 0.13310185814043507, |
| "epoch": 0.6470588235294118, |
| "frac_reward_zero_std": 0.4296875, |
| "grad_norm": 0.09454388235202577, |
| "learning_rate": 2.47716868444247e-06, |
| "loss": 0.0114, |
| "num_tokens": 51156518.0, |
| "reward": 0.4361979365348816, |
| "reward_std": 0.49607405066490173, |
| "rewards/reward_correctness/mean": 0.4361979067325592, |
| "rewards/reward_correctness/std": 0.4960741102695465, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999701976776123, |
| "sampling/importance_sampling_ratio/min": 0.02935052663087845, |
| "sampling/sampling_logp_difference/max": 3.528444766998291, |
| "sampling/sampling_logp_difference/mean": 0.011353356763720512, |
| "step": 44, |
| "step_time": 326.7605273928493 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0201822929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2984.0, |
| "completions/mean_length": 672.0306396484375, |
| "completions/mean_terminated_length": 622.5960083007812, |
| "completions/min_length": 143.0, |
| "completions/min_terminated_length": 143.0, |
| "entropy": 0.12670818017795682, |
| "epoch": 0.6617647058823529, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.08260192050872966, |
| "learning_rate": 2.4513471106295523e-06, |
| "loss": 0.0051, |
| "num_tokens": 52362397.0, |
| "reward": 0.4440104365348816, |
| "reward_std": 0.4970170855522156, |
| "rewards/reward_correctness/mean": 0.4440104067325592, |
| "rewards/reward_correctness/std": 0.49701711535453796, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000724792480469, |
| "sampling/importance_sampling_ratio/min": 0.09955490380525589, |
| "sampling/sampling_logp_difference/max": 2.3070459365844727, |
| "sampling/sampling_logp_difference/mean": 0.011110104620456696, |
| "step": 45, |
| "step_time": 311.18877177499235 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01953125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2785.0, |
| "completions/mean_length": 703.7669677734375, |
| "completions/mean_terminated_length": 656.5910034179688, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.12684163701487705, |
| "epoch": 0.6764705882352942, |
| "frac_reward_zero_std": 0.515625, |
| "grad_norm": 0.09331266808858216, |
| "learning_rate": 2.4250646895508992e-06, |
| "loss": 0.0032, |
| "num_tokens": 53619579.0, |
| "reward": 0.3912760615348816, |
| "reward_std": 0.48819488286972046, |
| "rewards/reward_correctness/mean": 0.3912760317325592, |
| "rewards/reward_correctness/std": 0.48819491267204285, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0001139640808105, |
| "sampling/importance_sampling_ratio/min": 0.06684955954551697, |
| "sampling/sampling_logp_difference/max": 2.705310583114624, |
| "sampling/sampling_logp_difference/mean": 0.011112010106444359, |
| "step": 46, |
| "step_time": 313.3187058418989 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0240885429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2982.0, |
| "completions/mean_length": 668.234375, |
| "completions/mean_terminated_length": 609.0033569335938, |
| "completions/min_length": 136.0, |
| "completions/min_terminated_length": 136.0, |
| "entropy": 0.11769095557974651, |
| "epoch": 0.6911764705882353, |
| "frac_reward_zero_std": 0.546875, |
| "grad_norm": 0.09763888479973193, |
| "learning_rate": 2.3983365359822804e-06, |
| "loss": 0.0065, |
| "num_tokens": 54811371.0, |
| "reward": 0.4407552182674408, |
| "reward_std": 0.49663931131362915, |
| "rewards/reward_correctness/mean": 0.4407552182674408, |
| "rewards/reward_correctness/std": 0.49663934111595154, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9966242909431458, |
| "sampling/importance_sampling_ratio/min": 1.1548664751366232e-17, |
| "sampling/sampling_logp_difference/max": 38.999961853027344, |
| "sampling/sampling_logp_difference/mean": 0.040139876306056976, |
| "step": 47, |
| "step_time": 321.21418457012624 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.021484375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2728.0, |
| "completions/mean_length": 687.427734375, |
| "completions/mean_terminated_length": 635.0718383789062, |
| "completions/min_length": 107.0, |
| "completions/min_terminated_length": 107.0, |
| "entropy": 0.12559914664598182, |
| "epoch": 0.7058823529411765, |
| "frac_reward_zero_std": 0.5390625, |
| "grad_norm": 0.09487698141413321, |
| "learning_rate": 2.3711780210360726e-06, |
| "loss": 0.0109, |
| "num_tokens": 56026512.0, |
| "reward": 0.3912760615348816, |
| "reward_std": 0.48819488286972046, |
| "rewards/reward_correctness/mean": 0.3912760317325592, |
| "rewards/reward_correctness/std": 0.48819491267204285, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000499486923218, |
| "sampling/importance_sampling_ratio/min": 1.1616037227213383e-05, |
| "sampling/sampling_logp_difference/max": 11.363123893737793, |
| "sampling/sampling_logp_difference/mean": 0.010995877906680107, |
| "step": 48, |
| "step_time": 309.3666119822301 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.017578125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2970.0, |
| "completions/mean_length": 698.48046875, |
| "completions/mean_terminated_length": 656.011962890625, |
| "completions/min_length": 157.0, |
| "completions/min_terminated_length": 157.0, |
| "entropy": 0.12552364665316418, |
| "epoch": 0.7205882352941176, |
| "frac_reward_zero_std": 0.484375, |
| "grad_norm": 0.09602299763561906, |
| "learning_rate": 2.343604763321476e-06, |
| "loss": 0.0123, |
| "num_tokens": 57251862.0, |
| "reward": 0.4134114682674408, |
| "reward_std": 0.49260571599006653, |
| "rewards/reward_correctness/mean": 0.4134114682674408, |
| "rewards/reward_correctness/std": 0.49260571599006653, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000723600387573, |
| "sampling/importance_sampling_ratio/min": 0.0009446050971746445, |
| "sampling/sampling_logp_difference/max": 6.964743614196777, |
| "sampling/sampling_logp_difference/mean": 0.010939620435237885, |
| "step": 49, |
| "step_time": 297.2958505139686 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0162760429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2870.0, |
| "completions/mean_length": 650.4967651367188, |
| "completions/mean_terminated_length": 610.43212890625, |
| "completions/min_length": 100.0, |
| "completions/min_terminated_length": 100.0, |
| "entropy": 0.11314014665549621, |
| "epoch": 0.7352941176470589, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.0970926202181468, |
| "learning_rate": 2.3156326199623965e-06, |
| "loss": 0.0121, |
| "num_tokens": 58399213.0, |
| "reward": 0.4583333432674408, |
| "reward_std": 0.49842309951782227, |
| "rewards/reward_correctness/mean": 0.4583333432674408, |
| "rewards/reward_correctness/std": 0.49842312932014465, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999960720539093, |
| "sampling/importance_sampling_ratio/min": 0.06841433048248291, |
| "sampling/sampling_logp_difference/max": 2.682173013687134, |
| "sampling/sampling_logp_difference/mean": 0.010027723386883736, |
| "step": 50, |
| "step_time": 305.2847441327758 |
| }, |
| { |
| "epoch": 0.7352941176470589, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.023809524519102915, |
| "eval_completions/max_length": 1246.345238095238, |
| "eval_completions/max_terminated_length": 1003.8809523809524, |
| "eval_completions/mean_length": 589.6845405215308, |
| "eval_completions/mean_terminated_length": 530.4769995553153, |
| "eval_completions/min_length": 251.0, |
| "eval_completions/min_terminated_length": 251.0, |
| "eval_entropy": 0.10575014485844544, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 58399213.0, |
| "eval_reward": 0.5000000120628447, |
| "eval_reward_std": 0.4889630675315857, |
| "eval_rewards/reward_correctness/mean": 0.5000000056766328, |
| "eval_rewards/reward_correctness/std": 0.4889630675315857, |
| "eval_runtime": 741.191, |
| "eval_samples_per_second": 0.675, |
| "eval_sampling/importance_sampling_ratio/max": 2.990457282179878, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9947371142251151, |
| "eval_sampling/importance_sampling_ratio/min": 0.5308067000338009, |
| "eval_sampling/sampling_logp_difference/max": 2.112378262338184, |
| "eval_sampling/sampling_logp_difference/mean": 0.027818603640688316, |
| "eval_steps_per_second": 0.113, |
| "step": 50 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01302083395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3064.0, |
| "completions/mean_length": 648.2448120117188, |
| "completions/mean_terminated_length": 616.2691650390625, |
| "completions/min_length": 143.0, |
| "completions/min_terminated_length": 143.0, |
| "entropy": 0.1220089640119113, |
| "epoch": 0.75, |
| "frac_reward_zero_std": 0.3984375, |
| "grad_norm": 0.10917632380343659, |
| "learning_rate": 2.2872776774781627e-06, |
| "loss": -0.0, |
| "num_tokens": 59534657.0, |
| "reward": 0.4407552182674408, |
| "reward_std": 0.49663931131362915, |
| "rewards/reward_correctness/mean": 0.4407552182674408, |
| "rewards/reward_correctness/std": 0.49663934111595154, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999938011169434, |
| "sampling/importance_sampling_ratio/min": 0.0287653636187315, |
| "sampling/sampling_logp_difference/max": 3.5485832691192627, |
| "sampling/sampling_logp_difference/mean": 0.010647393763065338, |
| "step": 51, |
| "step_time": 286.852661145851 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01302083395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3002.0, |
| "completions/mean_length": 652.33984375, |
| "completions/mean_terminated_length": 620.5264282226562, |
| "completions/min_length": 116.0, |
| "completions/min_terminated_length": 116.0, |
| "entropy": 0.11450612859334797, |
| "epoch": 0.7647058823529411, |
| "frac_reward_zero_std": 0.453125, |
| "grad_norm": 0.11777736125448364, |
| "learning_rate": 2.258556242532317e-06, |
| "loss": 0.0067, |
| "num_tokens": 60706655.0, |
| "reward": 0.4309895932674408, |
| "reward_std": 0.49537593126296997, |
| "rewards/reward_correctness/mean": 0.4309895932674408, |
| "rewards/reward_correctness/std": 0.49537593126296997, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9964901804924011, |
| "sampling/importance_sampling_ratio/min": 1.4169495366936997e-16, |
| "sampling/sampling_logp_difference/max": 36.492855072021484, |
| "sampling/sampling_logp_difference/mean": 0.04059432074427605, |
| "step": 52, |
| "step_time": 334.9317989842966 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.013671875, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2735.0, |
| "completions/mean_length": 650.4303588867188, |
| "completions/mean_terminated_length": 616.8640747070312, |
| "completions/min_length": 143.0, |
| "completions/min_terminated_length": 143.0, |
| "entropy": 0.12498974212212488, |
| "epoch": 0.7794117647058824, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.09996053573488595, |
| "learning_rate": 2.2294848325548066e-06, |
| "loss": 0.0031, |
| "num_tokens": 61861008.0, |
| "reward": 0.3795573115348816, |
| "reward_std": 0.48543480038642883, |
| "rewards/reward_correctness/mean": 0.3795572817325592, |
| "rewards/reward_correctness/std": 0.4854348301887512, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999858736991882, |
| "sampling/importance_sampling_ratio/min": 0.014357448555529118, |
| "sampling/sampling_logp_difference/max": 4.243486404418945, |
| "sampling/sampling_logp_difference/mean": 0.011073033325374126, |
| "step": 53, |
| "step_time": 285.01819738186896 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0162760429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3031.0, |
| "completions/mean_length": 631.8841552734375, |
| "completions/mean_terminated_length": 591.5115966796875, |
| "completions/min_length": 138.0, |
| "completions/min_terminated_length": 138.0, |
| "entropy": 0.10903365240665153, |
| "epoch": 0.7941176470588235, |
| "frac_reward_zero_std": 0.4375, |
| "grad_norm": 0.10210749738034491, |
| "learning_rate": 2.200080166242961e-06, |
| "loss": 0.0049, |
| "num_tokens": 62982650.0, |
| "reward": 0.4563802182674408, |
| "reward_std": 0.4982558488845825, |
| "rewards/reward_correctness/mean": 0.4563802182674408, |
| "rewards/reward_correctness/std": 0.4982558786869049, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999967098236084, |
| "sampling/importance_sampling_ratio/min": 0.011410914361476898, |
| "sampling/sampling_logp_difference/max": 4.473185062408447, |
| "sampling/sampling_logp_difference/mean": 0.009591399691998959, |
| "step": 54, |
| "step_time": 292.70618091057986 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2644.0, |
| "completions/mean_length": 683.1940307617188, |
| "completions/mean_terminated_length": 645.2764282226562, |
| "completions/min_length": 157.0, |
| "completions/min_terminated_length": 157.0, |
| "entropy": 0.1221256356802769, |
| "epoch": 0.8088235294117647, |
| "frac_reward_zero_std": 0.546875, |
| "grad_norm": 0.09445385336737874, |
| "learning_rate": 2.1703591539467283e-06, |
| "loss": 0.0127, |
| "num_tokens": 64207992.0, |
| "reward": 0.380859375, |
| "reward_std": 0.4857562184333801, |
| "rewards/reward_correctness/mean": 0.380859375, |
| "rewards/reward_correctness/std": 0.4857562482357025, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998800158500671, |
| "sampling/importance_sampling_ratio/min": 5.5687261919956654e-05, |
| "sampling/sampling_logp_difference/max": 9.795759201049805, |
| "sampling/sampling_logp_difference/mean": 0.010790005326271057, |
| "step": 55, |
| "step_time": 324.3310802578926 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0182291679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2970.0, |
| "completions/mean_length": 641.8776245117188, |
| "completions/mean_terminated_length": 596.8978881835938, |
| "completions/min_length": 89.0, |
| "completions/min_terminated_length": 89.0, |
| "entropy": 0.11542422254569829, |
| "epoch": 0.8235294117647058, |
| "frac_reward_zero_std": 0.5859375, |
| "grad_norm": 0.08674985428565889, |
| "learning_rate": 2.140338887943686e-06, |
| "loss": 0.0166, |
| "num_tokens": 65349976.0, |
| "reward": 0.47265625, |
| "reward_std": 0.499414324760437, |
| "rewards/reward_correctness/mean": 0.47265625, |
| "rewards/reward_correctness/std": 0.4994143545627594, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000611543655396, |
| "sampling/importance_sampling_ratio/min": 0.024185113608837128, |
| "sampling/sampling_logp_difference/max": 3.722018003463745, |
| "sampling/sampling_logp_difference/mean": 0.010218578390777111, |
| "step": 56, |
| "step_time": 296.37975679989904 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.010416666977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2546.0, |
| "completions/mean_length": 611.0390625, |
| "completions/mean_terminated_length": 585.1342163085938, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.11121287627611309, |
| "epoch": 0.8382352941176471, |
| "frac_reward_zero_std": 0.4296875, |
| "grad_norm": 0.10503693257109341, |
| "learning_rate": 2.110036632609435e-06, |
| "loss": 0.019, |
| "num_tokens": 66426784.0, |
| "reward": 0.4993489682674408, |
| "reward_std": 0.5001624226570129, |
| "rewards/reward_correctness/mean": 0.4993489682674408, |
| "rewards/reward_correctness/std": 0.5001624226570129, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999729990959167, |
| "sampling/importance_sampling_ratio/min": 0.0030124566983431578, |
| "sampling/sampling_logp_difference/max": 5.804999351501465, |
| "sampling/sampling_logp_difference/mean": 0.009916255250573158, |
| "step": 57, |
| "step_time": 296.1852462873794 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01692708395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3022.0, |
| "completions/mean_length": 698.658203125, |
| "completions/mean_terminated_length": 657.792724609375, |
| "completions/min_length": 129.0, |
| "completions/min_terminated_length": 129.0, |
| "entropy": 0.11595404433319345, |
| "epoch": 0.8529411764705882, |
| "frac_reward_zero_std": 0.4765625, |
| "grad_norm": 0.09551442970026573, |
| "learning_rate": 2.0794698144890156e-06, |
| "loss": 0.0119, |
| "num_tokens": 67683415.0, |
| "reward": 0.427734375, |
| "reward_std": 0.49491122364997864, |
| "rewards/reward_correctness/mean": 0.427734375, |
| "rewards/reward_correctness/std": 0.494911253452301, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9960780143737793, |
| "sampling/importance_sampling_ratio/min": 1.8462416315266454e-14, |
| "sampling/sampling_logp_difference/max": 31.62303924560547, |
| "sampling/sampling_logp_difference/mean": 0.043669816106557846, |
| "step": 58, |
| "step_time": 339.7408985039219 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01302083395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2949.0, |
| "completions/mean_length": 649.1393432617188, |
| "completions/mean_terminated_length": 617.1754760742188, |
| "completions/min_length": 137.0, |
| "completions/min_terminated_length": 137.0, |
| "entropy": 0.11917942861327901, |
| "epoch": 0.8676470588235294, |
| "frac_reward_zero_std": 0.546875, |
| "grad_norm": 0.10106429850918942, |
| "learning_rate": 2.048656012275064e-06, |
| "loss": -0.0005, |
| "num_tokens": 68856377.0, |
| "reward": 0.3697916865348816, |
| "reward_std": 0.48290538787841797, |
| "rewards/reward_correctness/mean": 0.3697916567325592, |
| "rewards/reward_correctness/std": 0.48290541768074036, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999811053276062, |
| "sampling/importance_sampling_ratio/min": 0.09501548111438751, |
| "sampling/sampling_logp_difference/max": 2.353715419769287, |
| "sampling/sampling_logp_difference/mean": 0.010620813816785812, |
| "step": 59, |
| "step_time": 312.87796821817756 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.029296875, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3066.0, |
| "completions/mean_length": 748.1400146484375, |
| "completions/mean_terminated_length": 678.0033569335938, |
| "completions/min_length": 123.0, |
| "completions/min_terminated_length": 123.0, |
| "entropy": 0.11793089081766084, |
| "epoch": 0.8823529411764706, |
| "frac_reward_zero_std": 0.5703125, |
| "grad_norm": 0.09465256634616137, |
| "learning_rate": 2.017612946698471e-06, |
| "loss": 0.0129, |
| "num_tokens": 70163368.0, |
| "reward": 0.3795573115348816, |
| "reward_std": 0.48543480038642883, |
| "rewards/reward_correctness/mean": 0.3795572817325592, |
| "rewards/reward_correctness/std": 0.4854348599910736, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999856948852539, |
| "sampling/importance_sampling_ratio/min": 0.016788704320788383, |
| "sampling/sampling_logp_difference/max": 4.0870490074157715, |
| "sampling/sampling_logp_difference/mean": 0.010650983080267906, |
| "step": 60, |
| "step_time": 309.0038933083415 |
| }, |
| { |
| "epoch": 0.8823529411764706, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.023809524519102915, |
| "eval_completions/max_length": 1401.5833333333333, |
| "eval_completions/max_terminated_length": 1113.4404761904761, |
| "eval_completions/mean_length": 614.5436699276879, |
| "eval_completions/mean_terminated_length": 554.2809684390113, |
| "eval_completions/min_length": 249.04761904761904, |
| "eval_completions/min_terminated_length": 249.04761904761904, |
| "eval_entropy": 0.10374800328697477, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 70163368.0, |
| "eval_reward": 0.5000000094019231, |
| "eval_reward_std": 0.4882907420396805, |
| "eval_rewards/reward_correctness/mean": 0.5000000058540276, |
| "eval_rewards/reward_correctness/std": 0.48829073849178495, |
| "eval_runtime": 832.1384, |
| "eval_samples_per_second": 0.601, |
| "eval_sampling/importance_sampling_ratio/max": 3.0, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9949375064600081, |
| "eval_sampling/importance_sampling_ratio/min": 0.5374763036767641, |
| "eval_sampling/sampling_logp_difference/max": 2.227688502697718, |
| "eval_sampling/sampling_logp_difference/mean": 0.027424226071508156, |
| "eval_steps_per_second": 0.101, |
| "step": 60 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0260416679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3042.0, |
| "completions/mean_length": 666.3509521484375, |
| "completions/mean_terminated_length": 602.0287475585938, |
| "completions/min_length": 140.0, |
| "completions/min_terminated_length": 140.0, |
| "entropy": 0.11709785764105618, |
| "epoch": 0.8970588235294118, |
| "frac_reward_zero_std": 0.5625, |
| "grad_norm": 0.09724587353427003, |
| "learning_rate": 1.9863584703373534e-06, |
| "loss": 0.0079, |
| "num_tokens": 71365839.0, |
| "reward": 0.41015625, |
| "reward_std": 0.4920220375061035, |
| "rewards/reward_correctness/mean": 0.41015625, |
| "rewards/reward_correctness/std": 0.4920220673084259, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9959919452667236, |
| "sampling/importance_sampling_ratio/min": 1.5554298034592538e-18, |
| "sampling/sampling_logp_difference/max": 41.00477981567383, |
| "sampling/sampling_logp_difference/mean": 0.044307515025138855, |
| "step": 61, |
| "step_time": 317.7361774868332 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0221354179084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3005.0, |
| "completions/mean_length": 715.015625, |
| "completions/mean_terminated_length": 661.6617431640625, |
| "completions/min_length": 154.0, |
| "completions/min_terminated_length": 154.0, |
| "entropy": 0.11619744420750067, |
| "epoch": 0.9117647058823529, |
| "frac_reward_zero_std": 0.4453125, |
| "grad_norm": 0.10403468878995832, |
| "learning_rate": 1.954910557350202e-06, |
| "loss": 0.0137, |
| "num_tokens": 72637971.0, |
| "reward": 0.3997395932674408, |
| "reward_std": 0.49000421166419983, |
| "rewards/reward_correctness/mean": 0.3997395932674408, |
| "rewards/reward_correctness/std": 0.4900042414665222, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000476837158203, |
| "sampling/importance_sampling_ratio/min": 0.028073355555534363, |
| "sampling/sampling_logp_difference/max": 3.57293438911438, |
| "sampling/sampling_logp_difference/mean": 0.010643940418958664, |
| "step": 62, |
| "step_time": 316.263129406143 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.014322916977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3012.0, |
| "completions/mean_length": 668.5345458984375, |
| "completions/mean_terminated_length": 633.6096801757812, |
| "completions/min_length": 151.0, |
| "completions/min_terminated_length": 151.0, |
| "entropy": 0.11588948150165379, |
| "epoch": 0.9264705882352942, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.10204530460670029, |
| "learning_rate": 1.9232872931391114e-06, |
| "loss": 0.0135, |
| "num_tokens": 73838996.0, |
| "reward": 0.3515625, |
| "reward_std": 0.4776136577129364, |
| "rewards/reward_correctness/mean": 0.3515625, |
| "rewards/reward_correctness/std": 0.4776136875152588, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999632239341736, |
| "sampling/importance_sampling_ratio/min": 0.0017267765942960978, |
| "sampling/sampling_logp_difference/max": 6.361498832702637, |
| "sampling/sampling_logp_difference/mean": 0.010555064305663109, |
| "step": 63, |
| "step_time": 296.9185470682569 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.015625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2800.0, |
| "completions/mean_length": 668.0931396484375, |
| "completions/mean_terminated_length": 629.9358520507812, |
| "completions/min_length": 97.0, |
| "completions/min_terminated_length": 97.0, |
| "entropy": 0.11724561610026285, |
| "epoch": 0.9411764705882353, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.09705948499787383, |
| "learning_rate": 1.8915068639490344e-06, |
| "loss": 0.009, |
| "num_tokens": 75027991.0, |
| "reward": 0.4264323115348816, |
| "reward_std": 0.4947192072868347, |
| "rewards/reward_correctness/mean": 0.4264322817325592, |
| "rewards/reward_correctness/std": 0.4947192370891571, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999983906745911, |
| "sampling/importance_sampling_ratio/min": 0.02149857021868229, |
| "sampling/sampling_logp_difference/max": 3.839768886566162, |
| "sampling/sampling_logp_difference/mean": 0.010617914609611034, |
| "step": 64, |
| "step_time": 313.9419138771482 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0201822929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2956.0, |
| "completions/mean_length": 672.75390625, |
| "completions/mean_terminated_length": 623.4869995117188, |
| "completions/min_length": 159.0, |
| "completions/min_terminated_length": 159.0, |
| "entropy": 0.11667637288337573, |
| "epoch": 0.9558823529411765, |
| "frac_reward_zero_std": 0.484375, |
| "grad_norm": 0.09957814700163607, |
| "learning_rate": 1.8595875464090389e-06, |
| "loss": 0.0099, |
| "num_tokens": 76231045.0, |
| "reward": 0.4661458432674408, |
| "reward_std": 0.4990150034427643, |
| "rewards/reward_correctness/mean": 0.4661458432674408, |
| "rewards/reward_correctness/std": 0.49901503324508667, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000202655792236, |
| "sampling/importance_sampling_ratio/min": 0.012628194876015186, |
| "sampling/sampling_logp_difference/max": 4.371823310852051, |
| "sampling/sampling_logp_difference/mean": 0.010494709014892578, |
| "step": 65, |
| "step_time": 301.15293512120843 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01888020895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2905.0, |
| "completions/mean_length": 651.5514526367188, |
| "completions/mean_terminated_length": 604.9734497070312, |
| "completions/min_length": 152.0, |
| "completions/min_terminated_length": 152.0, |
| "entropy": 0.11085655010538176, |
| "epoch": 0.9705882352941176, |
| "frac_reward_zero_std": 0.515625, |
| "grad_norm": 0.0997805532923189, |
| "learning_rate": 1.8275476970215906e-06, |
| "loss": 0.0033, |
| "num_tokens": 77403500.0, |
| "reward": 0.4140625, |
| "reward_std": 0.4927197992801666, |
| "rewards/reward_correctness/mean": 0.4140625, |
| "rewards/reward_correctness/std": 0.4927197992801666, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999725222587585, |
| "sampling/importance_sampling_ratio/min": 0.11216503381729126, |
| "sampling/sampling_logp_difference/max": 2.18778395652771, |
| "sampling/sampling_logp_difference/mean": 0.010185856372117996, |
| "step": 66, |
| "step_time": 305.88186718709767 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.012369791977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3066.0, |
| "completions/mean_length": 635.2025146484375, |
| "completions/mean_terminated_length": 604.6822509765625, |
| "completions/min_length": 111.0, |
| "completions/min_terminated_length": 111.0, |
| "entropy": 0.11111450218595564, |
| "epoch": 0.9852941176470589, |
| "frac_reward_zero_std": 0.4609375, |
| "grad_norm": 0.1004527266463445, |
| "learning_rate": 1.7954057416059002e-06, |
| "loss": 0.0031, |
| "num_tokens": 78520915.0, |
| "reward": 0.4296875, |
| "reward_std": 0.49519267678260803, |
| "rewards/reward_correctness/mean": 0.4296875, |
| "rewards/reward_correctness/std": 0.4951927065849304, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999885559082031, |
| "sampling/importance_sampling_ratio/min": 0.01389921735972166, |
| "sampling/sampling_logp_difference/max": 4.275922775268555, |
| "sampling/sampling_logp_difference/mean": 0.009849893860518932, |
| "step": 67, |
| "step_time": 289.09394181659445 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.006510416977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2954.0, |
| "completions/mean_length": 591.7311401367188, |
| "completions/mean_terminated_length": 575.4777221679688, |
| "completions/min_length": 171.0, |
| "completions/min_terminated_length": 171.0, |
| "entropy": 0.11097419285215437, |
| "epoch": 1.0, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.10381810303339536, |
| "learning_rate": 1.7631801647014034e-06, |
| "loss": 0.0062, |
| "num_tokens": 79582934.0, |
| "reward": 0.443359375, |
| "reward_std": 0.49694323539733887, |
| "rewards/reward_correctness/mean": 0.443359375, |
| "rewards/reward_correctness/std": 0.49694326519966125, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999539256095886, |
| "sampling/importance_sampling_ratio/min": 0.032169442623853683, |
| "sampling/sampling_logp_difference/max": 3.4367382526397705, |
| "sampling/sampling_logp_difference/mean": 0.009978776797652245, |
| "step": 68, |
| "step_time": 277.24001722969115 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02669270895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2924.0, |
| "completions/mean_length": 658.923828125, |
| "completions/mean_terminated_length": 592.8555297851562, |
| "completions/min_length": 112.0, |
| "completions/min_terminated_length": 112.0, |
| "entropy": 0.11762749508488923, |
| "epoch": 1.0147058823529411, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.09955363266557021, |
| "learning_rate": 1.7308894989374766e-06, |
| "loss": 0.0168, |
| "num_tokens": 80758973.0, |
| "reward": 0.404296875, |
| "reward_std": 0.4909152686595917, |
| "rewards/reward_correctness/mean": 0.404296875, |
| "rewards/reward_correctness/std": 0.49091529846191406, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000025987625122, |
| "sampling/importance_sampling_ratio/min": 0.06638438999652863, |
| "sampling/sampling_logp_difference/max": 2.7122933864593506, |
| "sampling/sampling_logp_difference/mean": 0.010275790467858315, |
| "step": 69, |
| "step_time": 303.03809165675193 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02473958395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2971.0, |
| "completions/mean_length": 651.80859375, |
| "completions/mean_terminated_length": 590.4152221679688, |
| "completions/min_length": 132.0, |
| "completions/min_terminated_length": 132.0, |
| "entropy": 0.10798334499122575, |
| "epoch": 1.0294117647058822, |
| "frac_reward_zero_std": 0.4921875, |
| "grad_norm": 0.09818298129014348, |
| "learning_rate": 1.6985523143754952e-06, |
| "loss": 0.0079, |
| "num_tokens": 81899183.0, |
| "reward": 0.4453125, |
| "reward_std": 0.4971621334552765, |
| "rewards/reward_correctness/mean": 0.4453125, |
| "rewards/reward_correctness/std": 0.4971621334552765, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000056028366089, |
| "sampling/importance_sampling_ratio/min": 0.0010057517793029547, |
| "sampling/sampling_logp_difference/max": 6.90201997756958, |
| "sampling/sampling_logp_difference/mean": 0.009786490350961685, |
| "step": 70, |
| "step_time": 286.29961240198463 |
| }, |
| { |
| "epoch": 1.0294117647058822, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.023809524519102915, |
| "eval_completions/max_length": 1280.6190476190477, |
| "eval_completions/max_terminated_length": 1000.1904761904761, |
| "eval_completions/mean_length": 570.4266034080869, |
| "eval_completions/mean_terminated_length": 510.04386247907365, |
| "eval_completions/min_length": 233.58333333333334, |
| "eval_completions/min_terminated_length": 233.58333333333334, |
| "eval_entropy": 0.09882747022701162, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 81899183.0, |
| "eval_reward": 0.513888899591707, |
| "eval_reward_std": 0.4880865791014263, |
| "eval_rewards/reward_correctness/mean": 0.5138888939150742, |
| "eval_rewards/reward_correctness/std": 0.4880865748439516, |
| "eval_runtime": 762.0449, |
| "eval_samples_per_second": 0.656, |
| "eval_sampling/importance_sampling_ratio/max": 2.9982360771724155, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9953853126083102, |
| "eval_sampling/importance_sampling_ratio/min": 0.5366758298838422, |
| "eval_sampling/sampling_logp_difference/max": 2.222159587201618, |
| "eval_sampling/sampling_logp_difference/mean": 0.026387135365179608, |
| "eval_steps_per_second": 0.11, |
| "step": 70 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.029296875, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2916.0, |
| "completions/mean_length": 677.2819213867188, |
| "completions/mean_terminated_length": 605.0067138671875, |
| "completions/min_length": 113.0, |
| "completions/min_terminated_length": 113.0, |
| "entropy": 0.11027820646995679, |
| "epoch": 1.0441176470588236, |
| "frac_reward_zero_std": 0.515625, |
| "grad_norm": 0.09778757444472032, |
| "learning_rate": 1.6661872078293582e-06, |
| "loss": 0.0019, |
| "num_tokens": 83093292.0, |
| "reward": 0.408203125, |
| "reward_std": 0.4916611611843109, |
| "rewards/reward_correctness/mean": 0.408203125, |
| "rewards/reward_correctness/std": 0.4916611909866333, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999990463256836, |
| "sampling/importance_sampling_ratio/min": 0.012234962545335293, |
| "sampling/sampling_logp_difference/max": 4.4034576416015625, |
| "sampling/sampling_logp_difference/mean": 0.010071806609630585, |
| "step": 71, |
| "step_time": 285.97908545611426 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0201822929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3048.0, |
| "completions/mean_length": 678.1654052734375, |
| "completions/mean_terminated_length": 628.8571166992188, |
| "completions/min_length": 133.0, |
| "completions/min_terminated_length": 133.0, |
| "entropy": 0.12014149565948173, |
| "epoch": 1.0588235294117647, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.0977574538696417, |
| "learning_rate": 1.6338127921706424e-06, |
| "loss": 0.0086, |
| "num_tokens": 84306890.0, |
| "reward": 0.3509114682674408, |
| "reward_std": 0.4774106740951538, |
| "rewards/reward_correctness/mean": 0.3509114682674408, |
| "rewards/reward_correctness/std": 0.4774107038974762, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999822974205017, |
| "sampling/importance_sampling_ratio/min": 1.1412609524086292e-07, |
| "sampling/sampling_logp_difference/max": 15.9859619140625, |
| "sampling/sampling_logp_difference/mean": 0.011068622581660748, |
| "step": 72, |
| "step_time": 297.21736324578524 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.012369791977107525, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3060.0, |
| "completions/mean_length": 608.6224365234375, |
| "completions/mean_terminated_length": 577.82861328125, |
| "completions/min_length": 71.0, |
| "completions/min_terminated_length": 71.0, |
| "entropy": 0.1028754398575984, |
| "epoch": 1.0735294117647058, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.11441655651882668, |
| "learning_rate": 1.6014476856245056e-06, |
| "loss": 0.0136, |
| "num_tokens": 85395286.0, |
| "reward": 0.4733073115348816, |
| "reward_std": 0.49944958090782166, |
| "rewards/reward_correctness/mean": 0.4733072817325592, |
| "rewards/reward_correctness/std": 0.49944961071014404, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000224113464355, |
| "sampling/importance_sampling_ratio/min": 5.602986675512511e-06, |
| "sampling/sampling_logp_difference/max": 12.09221076965332, |
| "sampling/sampling_logp_difference/mean": 0.009659310802817345, |
| "step": 73, |
| "step_time": 304.63233517389745 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0201822929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3070.0, |
| "completions/mean_length": 628.5150146484375, |
| "completions/mean_terminated_length": 578.1840209960938, |
| "completions/min_length": 96.0, |
| "completions/min_terminated_length": 96.0, |
| "entropy": 0.11295660870382562, |
| "epoch": 1.088235294117647, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.10090089187503627, |
| "learning_rate": 1.5691105010625233e-06, |
| "loss": 0.0161, |
| "num_tokens": 86510997.0, |
| "reward": 0.458984375, |
| "reward_std": 0.49847716093063354, |
| "rewards/reward_correctness/mean": 0.458984375, |
| "rewards/reward_correctness/std": 0.49847716093063354, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000795125961304, |
| "sampling/importance_sampling_ratio/min": 0.008855166845023632, |
| "sampling/sampling_logp_difference/max": 4.726754188537598, |
| "sampling/sampling_logp_difference/mean": 0.010539291426539421, |
| "step": 74, |
| "step_time": 302.7705778395757 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01888020895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2897.0, |
| "completions/mean_length": 601.7467651367188, |
| "completions/mean_terminated_length": 554.2103271484375, |
| "completions/min_length": 85.0, |
| "completions/min_terminated_length": 85.0, |
| "entropy": 0.10929441236658022, |
| "epoch": 1.1029411764705883, |
| "frac_reward_zero_std": 0.59375, |
| "grad_norm": 0.10200692192130892, |
| "learning_rate": 1.536819835298597e-06, |
| "loss": -0.0001, |
| "num_tokens": 87608464.0, |
| "reward": 0.484375, |
| "reward_std": 0.49991855025291443, |
| "rewards/reward_correctness/mean": 0.484375, |
| "rewards/reward_correctness/std": 0.49991855025291443, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9957535862922668, |
| "sampling/importance_sampling_ratio/min": 1.449367869865286e-16, |
| "sampling/sampling_logp_difference/max": 36.47023391723633, |
| "sampling/sampling_logp_difference/mean": 0.04288832098245621, |
| "step": 75, |
| "step_time": 311.1924736010842 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0234375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2985.0, |
| "completions/mean_length": 660.0189208984375, |
| "completions/mean_terminated_length": 602.13134765625, |
| "completions/min_length": 100.0, |
| "completions/min_terminated_length": 100.0, |
| "entropy": 0.1125782448798418, |
| "epoch": 1.1176470588235294, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.10450062970730634, |
| "learning_rate": 1.5045942583941002e-06, |
| "loss": 0.0069, |
| "num_tokens": 88781529.0, |
| "reward": 0.4166666865348816, |
| "reward_std": 0.4931671619415283, |
| "rewards/reward_correctness/mean": 0.4166666567325592, |
| "rewards/reward_correctness/std": 0.4931672215461731, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999983310699463, |
| "sampling/importance_sampling_ratio/min": 0.011418849229812622, |
| "sampling/sampling_logp_difference/max": 4.472489833831787, |
| "sampling/sampling_logp_difference/mean": 0.010256472043693066, |
| "step": 76, |
| "step_time": 301.73715012473986 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0182291679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3036.0, |
| "completions/mean_length": 644.708984375, |
| "completions/mean_terminated_length": 599.9515991210938, |
| "completions/min_length": 104.0, |
| "completions/min_terminated_length": 104.0, |
| "entropy": 0.1032532652898226, |
| "epoch": 1.1323529411764706, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.10228459845612378, |
| "learning_rate": 1.4724523029784097e-06, |
| "loss": 0.0058, |
| "num_tokens": 89906982.0, |
| "reward": 0.484375, |
| "reward_std": 0.49991855025291443, |
| "rewards/reward_correctness/mean": 0.484375, |
| "rewards/reward_correctness/std": 0.49991855025291443, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9998574256896973, |
| "sampling/importance_sampling_ratio/min": 0.0222180038690567, |
| "sampling/sampling_logp_difference/max": 3.806852340698242, |
| "sampling/sampling_logp_difference/mean": 0.009653305634856224, |
| "step": 77, |
| "step_time": 317.5641336161643 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0260416679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2996.0, |
| "completions/mean_length": 738.216796875, |
| "completions/mean_terminated_length": 675.816162109375, |
| "completions/min_length": 99.0, |
| "completions/min_terminated_length": 99.0, |
| "entropy": 0.10691302624763921, |
| "epoch": 1.1470588235294117, |
| "frac_reward_zero_std": 0.4765625, |
| "grad_norm": 0.09954934875420117, |
| "learning_rate": 1.4404124535909613e-06, |
| "loss": 0.0088, |
| "num_tokens": 91189983.0, |
| "reward": 0.43359375, |
| "reward_std": 0.49573197960853577, |
| "rewards/reward_correctness/mean": 0.43359375, |
| "rewards/reward_correctness/std": 0.49573197960853577, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999540448188782, |
| "sampling/importance_sampling_ratio/min": 0.006017227657139301, |
| "sampling/sampling_logp_difference/max": 5.113128662109375, |
| "sampling/sampling_logp_difference/mean": 0.009943715296685696, |
| "step": 78, |
| "step_time": 295.76658773003146 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.03125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2872.0, |
| "completions/mean_length": 699.9069213867188, |
| "completions/mean_terminated_length": 623.3877563476562, |
| "completions/min_length": 78.0, |
| "completions/min_terminated_length": 78.0, |
| "entropy": 0.10646833421196789, |
| "epoch": 1.161764705882353, |
| "frac_reward_zero_std": 0.4453125, |
| "grad_norm": 0.10932517333699619, |
| "learning_rate": 1.4084931360509656e-06, |
| "loss": 0.014, |
| "num_tokens": 92420284.0, |
| "reward": 0.5221354365348816, |
| "reward_std": 0.49967241287231445, |
| "rewards/reward_correctness/mean": 0.5221354365348816, |
| "rewards/reward_correctness/std": 0.49967247247695923, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000180006027222, |
| "sampling/importance_sampling_ratio/min": 0.025688737630844116, |
| "sampling/sampling_logp_difference/max": 3.6617026329040527, |
| "sampling/sampling_logp_difference/mean": 0.009844161570072174, |
| "step": 79, |
| "step_time": 317.65286510577425 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0279947929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2855.0, |
| "completions/mean_length": 709.154296875, |
| "completions/mean_terminated_length": 641.101806640625, |
| "completions/min_length": 135.0, |
| "completions/min_terminated_length": 135.0, |
| "entropy": 0.10584194795228541, |
| "epoch": 1.1764705882352942, |
| "frac_reward_zero_std": 0.5390625, |
| "grad_norm": 0.09227654592425696, |
| "learning_rate": 1.376712706860888e-06, |
| "loss": 0.0091, |
| "num_tokens": 93671677.0, |
| "reward": 0.3932291865348816, |
| "reward_std": 0.48862606287002563, |
| "rewards/reward_correctness/mean": 0.3932291567325592, |
| "rewards/reward_correctness/std": 0.488626092672348, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999242424964905, |
| "sampling/importance_sampling_ratio/min": 0.011996806599199772, |
| "sampling/sampling_logp_difference/max": 4.423114776611328, |
| "sampling/sampling_logp_difference/mean": 0.009747643955051899, |
| "step": 80, |
| "step_time": 302.8400612468831 |
| }, |
| { |
| "epoch": 1.1764705882352942, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.027777778605620067, |
| "eval_completions/max_length": 1370.3809523809523, |
| "eval_completions/max_terminated_length": 1022.8809523809524, |
| "eval_completions/mean_length": 592.8194589160737, |
| "eval_completions/mean_terminated_length": 521.4313621520996, |
| "eval_completions/min_length": 234.01190476190476, |
| "eval_completions/min_terminated_length": 234.01190476190476, |
| "eval_entropy": 0.0995181476076444, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 93671677.0, |
| "eval_reward": 0.5119047732580275, |
| "eval_reward_std": 0.5026668728817076, |
| "eval_rewards/reward_correctness/mean": 0.5119047675813947, |
| "eval_rewards/reward_correctness/std": 0.5026668714625495, |
| "eval_runtime": 818.4358, |
| "eval_samples_per_second": 0.611, |
| "eval_sampling/importance_sampling_ratio/max": 2.9984878642218455, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9950423446439561, |
| "eval_sampling/importance_sampling_ratio/min": 0.5204418023682332, |
| "eval_sampling/sampling_logp_difference/max": 2.3195427627790544, |
| "eval_sampling/sampling_logp_difference/mean": 0.026423310683596702, |
| "eval_steps_per_second": 0.103, |
| "step": 80 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.025390625, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2966.0, |
| "completions/mean_length": 725.4323120117188, |
| "completions/mean_terminated_length": 664.29931640625, |
| "completions/min_length": 115.0, |
| "completions/min_terminated_length": 115.0, |
| "entropy": 0.11524771642871201, |
| "epoch": 1.1911764705882353, |
| "frac_reward_zero_std": 0.578125, |
| "grad_norm": 0.09536117201504443, |
| "learning_rate": 1.3450894426497986e-06, |
| "loss": 0.002, |
| "num_tokens": 94931357.0, |
| "reward": 0.3997395932674408, |
| "reward_std": 0.49000421166419983, |
| "rewards/reward_correctness/mean": 0.3997395932674408, |
| "rewards/reward_correctness/std": 0.4900042414665222, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000006079673767, |
| "sampling/importance_sampling_ratio/min": 0.018972357735037804, |
| "sampling/sampling_logp_difference/max": 3.9647722244262695, |
| "sampling/sampling_logp_difference/mean": 0.010577967390418053, |
| "step": 81, |
| "step_time": 311.70652225101367 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02734375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2805.0, |
| "completions/mean_length": 665.19140625, |
| "completions/mean_terminated_length": 597.5300903320312, |
| "completions/min_length": 136.0, |
| "completions/min_terminated_length": 136.0, |
| "entropy": 0.10638129588915035, |
| "epoch": 1.2058823529411764, |
| "frac_reward_zero_std": 0.453125, |
| "grad_norm": 0.10873786908331745, |
| "learning_rate": 1.313641529662647e-06, |
| "loss": 0.014, |
| "num_tokens": 96100355.0, |
| "reward": 0.4505208432674408, |
| "reward_std": 0.4977077841758728, |
| "rewards/reward_correctness/mean": 0.4505208432674408, |
| "rewards/reward_correctness/std": 0.4977078139781952, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999487400054932, |
| "sampling/importance_sampling_ratio/min": 0.003440916072577238, |
| "sampling/sampling_logp_difference/max": 5.672017574310303, |
| "sampling/sampling_logp_difference/mean": 0.009805039502680302, |
| "step": 82, |
| "step_time": 290.32841725600883 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0162760429084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2964.0, |
| "completions/mean_length": 650.9525146484375, |
| "completions/mean_terminated_length": 611.2316284179688, |
| "completions/min_length": 130.0, |
| "completions/min_terminated_length": 130.0, |
| "entropy": 0.10257759928936139, |
| "epoch": 1.2205882352941178, |
| "frac_reward_zero_std": 0.5078125, |
| "grad_norm": 0.10011774111650357, |
| "learning_rate": 1.2823870533015295e-06, |
| "loss": 0.0071, |
| "num_tokens": 97261378.0, |
| "reward": 0.4739583432674408, |
| "reward_std": 0.49948397278785706, |
| "rewards/reward_correctness/mean": 0.4739583432674408, |
| "rewards/reward_correctness/std": 0.49948397278785706, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999834895133972, |
| "sampling/importance_sampling_ratio/min": 0.051053013652563095, |
| "sampling/sampling_logp_difference/max": 3.1086947917938232, |
| "sampling/sampling_logp_difference/mean": 0.009559324942529202, |
| "step": 83, |
| "step_time": 314.8325279019773 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01888020895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3036.0, |
| "completions/mean_length": 677.6198120117188, |
| "completions/mean_terminated_length": 631.54345703125, |
| "completions/min_length": 134.0, |
| "completions/min_terminated_length": 134.0, |
| "entropy": 0.11533306317869574, |
| "epoch": 1.2352941176470589, |
| "frac_reward_zero_std": 0.5546875, |
| "grad_norm": 0.09282478295845689, |
| "learning_rate": 1.2513439877249363e-06, |
| "loss": 0.004, |
| "num_tokens": 98480102.0, |
| "reward": 0.4225260615348816, |
| "reward_std": 0.49412214756011963, |
| "rewards/reward_correctness/mean": 0.4225260317325592, |
| "rewards/reward_correctness/std": 0.4941222369670868, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000012993812561, |
| "sampling/importance_sampling_ratio/min": 0.051133543252944946, |
| "sampling/sampling_logp_difference/max": 2.9733145236968994, |
| "sampling/sampling_logp_difference/mean": 0.01072630099952221, |
| "step": 84, |
| "step_time": 317.55150324339047 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.03059895895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2932.0, |
| "completions/mean_length": 699.4720458984375, |
| "completions/mean_terminated_length": 624.5836181640625, |
| "completions/min_length": 120.0, |
| "completions/min_terminated_length": 120.0, |
| "entropy": 0.11002266616560519, |
| "epoch": 1.25, |
| "frac_reward_zero_std": 0.4921875, |
| "grad_norm": 0.09594840299692999, |
| "learning_rate": 1.220530185510985e-06, |
| "loss": 0.0126, |
| "num_tokens": 99716383.0, |
| "reward": 0.4700520932674408, |
| "reward_std": 0.49926483631134033, |
| "rewards/reward_correctness/mean": 0.4700520932674408, |
| "rewards/reward_correctness/std": 0.4992648661136627, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999629855155945, |
| "sampling/importance_sampling_ratio/min": 0.042671963572502136, |
| "sampling/sampling_logp_difference/max": 3.1542131900787354, |
| "sampling/sampling_logp_difference/mean": 0.010363861918449402, |
| "step": 85, |
| "step_time": 306.6277994834818 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02473958395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3045.0, |
| "completions/mean_length": 699.0983276367188, |
| "completions/mean_terminated_length": 638.904541015625, |
| "completions/min_length": 91.0, |
| "completions/min_terminated_length": 91.0, |
| "entropy": 0.11227407562546432, |
| "epoch": 1.2647058823529411, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.0976507340489589, |
| "learning_rate": 1.189963367390565e-06, |
| "loss": 0.0028, |
| "num_tokens": 100929470.0, |
| "reward": 0.4270833432674408, |
| "reward_std": 0.4948156476020813, |
| "rewards/reward_correctness/mean": 0.4270833432674408, |
| "rewards/reward_correctness/std": 0.4948156774044037, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000653266906738, |
| "sampling/importance_sampling_ratio/min": 0.00828573852777481, |
| "sampling/sampling_logp_difference/max": 4.793219566345215, |
| "sampling/sampling_logp_difference/mean": 0.010349968448281288, |
| "step": 86, |
| "step_time": 309.40160133549944 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.033203125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3067.0, |
| "completions/mean_length": 759.8939208984375, |
| "completions/mean_terminated_length": 680.816162109375, |
| "completions/min_length": 143.0, |
| "completions/min_terminated_length": 143.0, |
| "entropy": 0.10843936848687008, |
| "epoch": 1.2794117647058822, |
| "frac_reward_zero_std": 0.5390625, |
| "grad_norm": 0.089055654394311, |
| "learning_rate": 1.159661112056314e-06, |
| "loss": 0.0081, |
| "num_tokens": 102291799.0, |
| "reward": 0.3932291865348816, |
| "reward_std": 0.48862606287002563, |
| "rewards/reward_correctness/mean": 0.3932291567325592, |
| "rewards/reward_correctness/std": 0.488626092672348, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999238848686218, |
| "sampling/importance_sampling_ratio/min": 0.0069398460909724236, |
| "sampling/sampling_logp_difference/max": 4.970475673675537, |
| "sampling/sampling_logp_difference/mean": 0.009973037987947464, |
| "step": 87, |
| "step_time": 333.1185615430586 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0234375, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3020.0, |
| "completions/mean_length": 676.6627807617188, |
| "completions/mean_terminated_length": 619.1746826171875, |
| "completions/min_length": 161.0, |
| "completions/min_terminated_length": 161.0, |
| "entropy": 0.10631966090295464, |
| "epoch": 1.2941176470588236, |
| "frac_reward_zero_std": 0.4609375, |
| "grad_norm": 0.10902086573519744, |
| "learning_rate": 1.1296408460532715e-06, |
| "loss": 0.0173, |
| "num_tokens": 103498073.0, |
| "reward": 0.46875, |
| "reward_std": 0.4991849958896637, |
| "rewards/reward_correctness/mean": 0.46875, |
| "rewards/reward_correctness/std": 0.4991849958896637, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999716877937317, |
| "sampling/importance_sampling_ratio/min": 0.04054650664329529, |
| "sampling/sampling_logp_difference/max": 3.205305576324463, |
| "sampling/sampling_logp_difference/mean": 0.00969459768384695, |
| "step": 88, |
| "step_time": 307.3499052450061 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0279947929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2906.0, |
| "completions/mean_length": 706.0625, |
| "completions/mean_terminated_length": 637.9209594726562, |
| "completions/min_length": 81.0, |
| "completions/min_terminated_length": 81.0, |
| "entropy": 0.10058472969103605, |
| "epoch": 1.3088235294117647, |
| "frac_reward_zero_std": 0.546875, |
| "grad_norm": 0.09902012748669947, |
| "learning_rate": 1.0999198337570392e-06, |
| "loss": 0.0137, |
| "num_tokens": 104734205.0, |
| "reward": 0.4596354365348816, |
| "reward_std": 0.4985303282737732, |
| "rewards/reward_correctness/mean": 0.4596354067325592, |
| "rewards/reward_correctness/std": 0.4985303580760956, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999986290931702, |
| "sampling/importance_sampling_ratio/min": 7.468889816664159e-05, |
| "sampling/sampling_logp_difference/max": 9.502179145812988, |
| "sampling/sampling_logp_difference/mean": 0.00929464865475893, |
| "step": 89, |
| "step_time": 299.34720146795735 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.01953125, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2999.0, |
| "completions/mean_length": 693.0866088867188, |
| "completions/mean_terminated_length": 645.9681396484375, |
| "completions/min_length": 155.0, |
| "completions/min_terminated_length": 155.0, |
| "entropy": 0.10605936858337373, |
| "epoch": 1.3235294117647058, |
| "frac_reward_zero_std": 0.53125, |
| "grad_norm": 0.09893613436677237, |
| "learning_rate": 1.0705151674451938e-06, |
| "loss": 0.0013, |
| "num_tokens": 105954954.0, |
| "reward": 0.412109375, |
| "reward_std": 0.49237489700317383, |
| "rewards/reward_correctness/mean": 0.412109375, |
| "rewards/reward_correctness/std": 0.4923749268054962, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000171661376953, |
| "sampling/importance_sampling_ratio/min": 0.0009838553378358483, |
| "sampling/sampling_logp_difference/max": 6.924031734466553, |
| "sampling/sampling_logp_difference/mean": 0.010026591829955578, |
| "step": 90, |
| "step_time": 320.5724582099356 |
| }, |
| { |
| "epoch": 1.3235294117647058, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.01984127043258576, |
| "eval_completions/max_length": 1227.952380952381, |
| "eval_completions/max_terminated_length": 1012.6428571428571, |
| "eval_completions/mean_length": 574.4107317243304, |
| "eval_completions/mean_terminated_length": 524.4579500470843, |
| "eval_completions/min_length": 231.54761904761904, |
| "eval_completions/min_terminated_length": 231.54761904761904, |
| "eval_entropy": 0.09793427143068541, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 105954954.0, |
| "eval_reward": 0.5218254084743205, |
| "eval_reward_std": 0.49357138325770694, |
| "eval_rewards/reward_correctness/mean": 0.5218254020881086, |
| "eval_rewards/reward_correctness/std": 0.4935713772262846, |
| "eval_runtime": 727.7944, |
| "eval_samples_per_second": 0.687, |
| "eval_sampling/importance_sampling_ratio/max": 2.9972963560195196, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9947727024555206, |
| "eval_sampling/importance_sampling_ratio/min": 0.5220564539943423, |
| "eval_sampling/sampling_logp_difference/max": 2.271739939848582, |
| "eval_sampling/sampling_logp_difference/mean": 0.025922782874355715, |
| "eval_steps_per_second": 0.115, |
| "step": 90 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02669270895421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2971.0, |
| "completions/mean_length": 670.478515625, |
| "completions/mean_terminated_length": 604.6173706054688, |
| "completions/min_length": 168.0, |
| "completions/min_terminated_length": 168.0, |
| "entropy": 0.11070733214728534, |
| "epoch": 1.3382352941176472, |
| "frac_reward_zero_std": 0.46875, |
| "grad_norm": 0.10384112383517705, |
| "learning_rate": 1.0414437574676832e-06, |
| "loss": 0.0096, |
| "num_tokens": 107175105.0, |
| "reward": 0.4850260615348816, |
| "reward_std": 0.4999384880065918, |
| "rewards/reward_correctness/mean": 0.4850260317325592, |
| "rewards/reward_correctness/std": 0.4999385178089142, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999462962150574, |
| "sampling/importance_sampling_ratio/min": 0.015210184268653393, |
| "sampling/sampling_logp_difference/max": 4.185790061950684, |
| "sampling/sampling_logp_difference/mean": 0.010488501749932766, |
| "step": 91, |
| "step_time": 318.08187717711553 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0572916679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2974.0, |
| "completions/mean_length": 771.5859375, |
| "completions/mean_terminated_length": 632.3377075195312, |
| "completions/min_length": 138.0, |
| "completions/min_terminated_length": 138.0, |
| "entropy": 0.1037912426982075, |
| "epoch": 1.3529411764705883, |
| "frac_reward_zero_std": 0.5234375, |
| "grad_norm": 0.09178918055813894, |
| "learning_rate": 1.0127223225218379e-06, |
| "loss": 0.0206, |
| "num_tokens": 108526425.0, |
| "reward": 0.4505208432674408, |
| "reward_std": 0.4977077841758728, |
| "rewards/reward_correctness/mean": 0.4505208432674408, |
| "rewards/reward_correctness/std": 0.4977078139781952, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000383853912354, |
| "sampling/importance_sampling_ratio/min": 0.0019341822480782866, |
| "sampling/sampling_logp_difference/max": 6.24807071685791, |
| "sampling/sampling_logp_difference/mean": 0.009867710061371326, |
| "step": 92, |
| "step_time": 316.0852893306874 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02473958395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3047.0, |
| "completions/mean_length": 672.578125, |
| "completions/mean_terminated_length": 611.8264770507812, |
| "completions/min_length": 136.0, |
| "completions/min_terminated_length": 136.0, |
| "entropy": 0.10452087642624974, |
| "epoch": 1.3676470588235294, |
| "frac_reward_zero_std": 0.625, |
| "grad_norm": 0.0979449353074455, |
| "learning_rate": 9.843673800376037e-07, |
| "loss": 0.0097, |
| "num_tokens": 109731417.0, |
| "reward": 0.5006510615348816, |
| "reward_std": 0.5001624226570129, |
| "rewards/reward_correctness/mean": 0.5006510615348816, |
| "rewards/reward_correctness/std": 0.5001624226570129, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9955242276191711, |
| "sampling/importance_sampling_ratio/min": 2.6044188559093452e-17, |
| "sampling/sampling_logp_difference/max": 38.186737060546875, |
| "sampling/sampling_logp_difference/mean": 0.04697582125663757, |
| "step": 93, |
| "step_time": 322.82180776353925 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0182291679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2877.0, |
| "completions/mean_length": 635.4290771484375, |
| "completions/mean_terminated_length": 590.1876831054688, |
| "completions/min_length": 97.0, |
| "completions/min_terminated_length": 97.0, |
| "entropy": 0.10279296265798621, |
| "epoch": 1.3823529411764706, |
| "frac_reward_zero_std": 0.546875, |
| "grad_norm": 0.10426534258129598, |
| "learning_rate": 9.563952366785246e-07, |
| "loss": 0.005, |
| "num_tokens": 110869904.0, |
| "reward": 0.453125, |
| "reward_std": 0.4979599714279175, |
| "rewards/reward_correctness/mean": 0.453125, |
| "rewards/reward_correctness/std": 0.49796000123023987, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.000075340270996, |
| "sampling/importance_sampling_ratio/min": 0.010155136696994305, |
| "sampling/sampling_logp_difference/max": 4.589775562286377, |
| "sampling/sampling_logp_difference/mean": 0.009832991287112236, |
| "step": 94, |
| "step_time": 315.47392428759485 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0338541679084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3048.0, |
| "completions/mean_length": 740.6784057617188, |
| "completions/mean_terminated_length": 659.07275390625, |
| "completions/min_length": 123.0, |
| "completions/min_terminated_length": 123.0, |
| "entropy": 0.10438565909862518, |
| "epoch": 1.3970588235294117, |
| "frac_reward_zero_std": 0.46875, |
| "grad_norm": 0.10174293989185926, |
| "learning_rate": 9.288219789639276e-07, |
| "loss": 0.0078, |
| "num_tokens": 112162002.0, |
| "reward": 0.3971354365348816, |
| "reward_std": 0.48946380615234375, |
| "rewards/reward_correctness/mean": 0.3971354067325592, |
| "rewards/reward_correctness/std": 0.4894638657569885, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000094175338745, |
| "sampling/importance_sampling_ratio/min": 0.05274541676044464, |
| "sampling/sampling_logp_difference/max": 2.9422783851623535, |
| "sampling/sampling_logp_difference/mean": 0.009899480268359184, |
| "step": 95, |
| "step_time": 319.02526410855353 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0279947929084301, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3001.0, |
| "completions/mean_length": 696.400390625, |
| "completions/mean_terminated_length": 627.9805908203125, |
| "completions/min_length": 112.0, |
| "completions/min_terminated_length": 112.0, |
| "entropy": 0.1095525169512257, |
| "epoch": 1.4117647058823528, |
| "frac_reward_zero_std": 0.4921875, |
| "grad_norm": 0.09956779342472952, |
| "learning_rate": 9.016634640177203e-07, |
| "loss": 0.0032, |
| "num_tokens": 113383389.0, |
| "reward": 0.4700520932674408, |
| "reward_std": 0.49926483631134033, |
| "rewards/reward_correctness/mean": 0.4700520932674408, |
| "rewards/reward_correctness/std": 0.4992648661136627, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.999976634979248, |
| "sampling/importance_sampling_ratio/min": 0.018746379762887955, |
| "sampling/sampling_logp_difference/max": 3.976754665374756, |
| "sampling/sampling_logp_difference/mean": 0.010300280526280403, |
| "step": 96, |
| "step_time": 296.686714746058 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.0442708358168602, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2918.0, |
| "completions/mean_length": 707.390625, |
| "completions/mean_terminated_length": 597.8582763671875, |
| "completions/min_length": 109.0, |
| "completions/min_terminated_length": 109.0, |
| "entropy": 0.10301207873271778, |
| "epoch": 1.4264705882352942, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.10752603822364419, |
| "learning_rate": 8.74935310449101e-07, |
| "loss": 0.0101, |
| "num_tokens": 114641661.0, |
| "reward": 0.4459635615348816, |
| "reward_std": 0.4972333610057831, |
| "rewards/reward_correctness/mean": 0.4459635317325592, |
| "rewards/reward_correctness/std": 0.49723339080810547, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9956812858581543, |
| "sampling/importance_sampling_ratio/min": 1.156828547549599e-17, |
| "sampling/sampling_logp_difference/max": 38.99826431274414, |
| "sampling/sampling_logp_difference/mean": 0.05035912245512009, |
| "step": 97, |
| "step_time": 362.47534032771364 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02473958395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 2888.0, |
| "completions/mean_length": 693.869140625, |
| "completions/mean_terminated_length": 633.542724609375, |
| "completions/min_length": 138.0, |
| "completions/min_terminated_length": 138.0, |
| "entropy": 0.10327525227330625, |
| "epoch": 1.4411764705882353, |
| "frac_reward_zero_std": 0.453125, |
| "grad_norm": 0.11138329626839688, |
| "learning_rate": 8.486528893704481e-07, |
| "loss": 0.0123, |
| "num_tokens": 115854972.0, |
| "reward": 0.4765625, |
| "reward_std": 0.4996130168437958, |
| "rewards/reward_correctness/mean": 0.4765625, |
| "rewards/reward_correctness/std": 0.49961304664611816, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000249147415161, |
| "sampling/importance_sampling_ratio/min": 0.003647557459771633, |
| "sampling/sampling_logp_difference/max": 5.613697528839111, |
| "sampling/sampling_logp_difference/mean": 0.009877603501081467, |
| "step": 98, |
| "step_time": 316.61871033906937 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02473958395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3030.0, |
| "completions/mean_length": 681.708984375, |
| "completions/mean_terminated_length": 621.0740966796875, |
| "completions/min_length": 94.0, |
| "completions/min_terminated_length": 94.0, |
| "entropy": 0.09731233183993027, |
| "epoch": 1.4558823529411764, |
| "frac_reward_zero_std": 0.5390625, |
| "grad_norm": 0.09440221134197979, |
| "learning_rate": 8.228313155575304e-07, |
| "loss": 0.0093, |
| "num_tokens": 117048021.0, |
| "reward": 0.4459635615348816, |
| "reward_std": 0.4972333312034607, |
| "rewards/reward_correctness/mean": 0.4459635317325592, |
| "rewards/reward_correctness/std": 0.49723339080810547, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 1.0000076293945312, |
| "sampling/importance_sampling_ratio/min": 0.025448348373174667, |
| "sampling/sampling_logp_difference/max": 3.6711044311523438, |
| "sampling/sampling_logp_difference/mean": 0.0093543641269207, |
| "step": 99, |
| "step_time": 289.0866700396873 |
| }, |
| { |
| "clip_ratio/high_max": 0.0, |
| "clip_ratio/high_mean": 0.0, |
| "clip_ratio/low_mean": 0.0, |
| "clip_ratio/low_min": 0.0, |
| "clip_ratio/region_mean": 0.0, |
| "completions/clipped_ratio": 0.02864583395421505, |
| "completions/max_length": 3072.0, |
| "completions/max_terminated_length": 3048.0, |
| "completions/mean_length": 721.3880615234375, |
| "completions/mean_terminated_length": 652.0676879882812, |
| "completions/min_length": 84.0, |
| "completions/min_terminated_length": 84.0, |
| "entropy": 0.10687656450318173, |
| "epoch": 1.4705882352941178, |
| "frac_reward_zero_std": 0.5, |
| "grad_norm": 0.10480703467405905, |
| "learning_rate": 7.97485438757144e-07, |
| "loss": 0.0068, |
| "num_tokens": 118314965.0, |
| "reward": 0.404296875, |
| "reward_std": 0.4909152686595917, |
| "rewards/reward_correctness/mean": 0.404296875, |
| "rewards/reward_correctness/std": 0.49091529846191406, |
| "sampling/importance_sampling_ratio/max": 3.0, |
| "sampling/importance_sampling_ratio/mean": 0.9999392628669739, |
| "sampling/importance_sampling_ratio/min": 0.018848691135644913, |
| "sampling/sampling_logp_difference/max": 3.9713118076324463, |
| "sampling/sampling_logp_difference/mean": 0.01010206714272499, |
| "step": 100, |
| "step_time": 316.2152795935981 |
| }, |
| { |
| "epoch": 1.4705882352941178, |
| "eval_clip_ratio/high_max": 0.0, |
| "eval_clip_ratio/high_mean": 0.0, |
| "eval_clip_ratio/low_mean": 0.0, |
| "eval_clip_ratio/low_min": 0.0, |
| "eval_clip_ratio/region_mean": 0.0, |
| "eval_completions/clipped_ratio": 0.029761905648878643, |
| "eval_completions/max_length": 1368.107142857143, |
| "eval_completions/max_terminated_length": 995.8095238095239, |
| "eval_completions/mean_length": 603.3313642229352, |
| "eval_completions/mean_terminated_length": 527.4833472115653, |
| "eval_completions/min_length": 238.27380952380952, |
| "eval_completions/min_terminated_length": 238.27380952380952, |
| "eval_entropy": 0.09634455040629421, |
| "eval_frac_reward_zero_std": 1.0, |
| "eval_loss": 0.0, |
| "eval_num_tokens": 118314965.0, |
| "eval_reward": 0.52976191646996, |
| "eval_reward_std": 0.4925737629334132, |
| "eval_rewards/reward_correctness/mean": 0.5297619107933271, |
| "eval_rewards/reward_correctness/std": 0.4925737593855177, |
| "eval_runtime": 817.1205, |
| "eval_samples_per_second": 0.612, |
| "eval_sampling/importance_sampling_ratio/max": 2.998494170960926, |
| "eval_sampling/importance_sampling_ratio/mean": 0.9951962956360408, |
| "eval_sampling/importance_sampling_ratio/min": 0.5286199062885273, |
| "eval_sampling/sampling_logp_difference/max": 2.220592745712825, |
| "eval_sampling/sampling_logp_difference/mean": 0.02555376207012506, |
| "eval_steps_per_second": 0.103, |
| "step": 100 |
| } |
| ], |
| "logging_steps": 1.0, |
| "max_steps": 136, |
| "num_input_tokens_seen": 118314965, |
| "num_train_epochs": 2, |
| "save_steps": 10, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 0.0, |
| "train_batch_size": 2, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|