{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.975609756097561, "eval_steps": 500, "global_step": 40, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9958333333333333, "completions/max_length": 256.0, "completions/max_terminated_length": 88.4, "completions/mean_length": 255.9635437011719, "completions/mean_terminated_length": 88.4, "completions/min_length": 242.0, "completions/min_terminated_length": 88.4, "epoch": 0.24390243902439024, "frac_reward_zero_std": 0.0, "grad_norm": 0.3228670060634613, "kl": 0.0005886403494514525, "learning_rate": 1.3802524773477098e-06, "loss": 0.0001, "num_tokens": 540050.0, "reward": 0.33145859837532043, "reward_std": 0.026524317637085914, "rewards/wrapper/mean": 0.165729296207428, "rewards/wrapper/std": 0.019212284684181215, "step": 5 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9958333333333333, "completions/max_length": 256.0, "completions/max_terminated_length": 84.6, "completions/mean_length": 255.95364685058593, "completions/mean_terminated_length": 84.6, "completions/min_length": 238.2, "completions/min_terminated_length": 84.6, "epoch": 0.4878048780487805, "frac_reward_zero_std": 0.0010416666977107526, "grad_norm": 0.2965202331542969, "kl": 0.0011322355625452475, "learning_rate": 3.105568074032347e-06, "loss": 0.0001, "num_tokens": 1079737.0, "reward": 0.3324328601360321, "reward_std": 0.028421235084533692, "rewards/wrapper/mean": 0.16621642410755158, "rewards/wrapper/std": 0.019593635573983192, "step": 10 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "epoch": 0.7317073170731707, "frac_reward_zero_std": 0.0031250000931322573, "grad_norm": 0.284412145614624, "kl": 0.0018224636791273952, "learning_rate": 4.830883670716985e-06, "loss": 0.0001, "num_tokens": 1618911.0, "reward": 0.3351516366004944, "reward_std": 0.027410299330949784, "rewards/wrapper/mean": 0.1675758183002472, "rewards/wrapper/std": 0.019275682047009468, "step": 15 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.9979166666666668, "completions/max_length": 256.0, "completions/max_terminated_length": 40.6, "completions/mean_length": 255.97239685058594, "completions/mean_terminated_length": 40.6, "completions/min_length": 245.4, "completions/min_terminated_length": 40.6, "epoch": 0.975609756097561, "frac_reward_zero_std": 0.0, "grad_norm": 0.3010239601135254, "kl": 0.003081864328123629, "learning_rate": 6.556199267401621e-06, "loss": 0.0002, "num_tokens": 2158306.0, "reward": 0.3369597256183624, "reward_std": 0.029058777168393134, "rewards/wrapper/mean": 0.16847985982894897, "rewards/wrapper/std": 0.01961274929344654, "step": 20 }, { "epoch": 0.975609756097561, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 1.0, "eval_completions/max_length": 256.0, "eval_completions/max_terminated_length": 0.0, "eval_completions/mean_length": 256.0, "eval_completions/mean_terminated_length": 0.0, "eval_completions/min_length": 256.0, "eval_completions/min_terminated_length": 0.0, "eval_frac_reward_zero_std": 0.0, "eval_kl": 0.004834831831976772, "eval_loss": 0.00019452218839433044, "eval_num_tokens": 2158306.0, "eval_reward": 0.3390903663635254, "eval_reward_std": 0.025529487021267415, "eval_rewards/wrapper/mean": 0.1695451831817627, "eval_rewards/wrapper/std": 0.017574026770889758, "eval_runtime": 76.535, "eval_samples_per_second": 2.613, "eval_steps_per_second": 0.17, "step": 20 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "epoch": 1.2439024390243902, "frac_reward_zero_std": 0.0, "grad_norm": 0.3230460584163666, "kl": 0.007101990515366197, "learning_rate": 8.281514864086258e-06, "loss": 0.0003, "num_tokens": 2698886.0, "reward": 0.34189478754997255, "reward_std": 0.02718120105564594, "rewards/wrapper/mean": 0.1709473878145218, "rewards/wrapper/std": 0.019510553032159806, "step": 25 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "epoch": 1.4878048780487805, "frac_reward_zero_std": 0.002083333395421505, "grad_norm": 0.29992207884788513, "kl": 0.015912340488284826, "learning_rate": 1.0006830460770896e-05, "loss": 0.0006, "num_tokens": 3238296.0, "reward": 0.34575957655906675, "reward_std": 0.028136394917964935, "rewards/wrapper/mean": 0.17287977933883666, "rewards/wrapper/std": 0.019339724257588387, "step": 30 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "epoch": 1.7317073170731707, "frac_reward_zero_std": 0.0010416666977107526, "grad_norm": 0.2933114469051361, "kl": 0.02282416820526123, "learning_rate": 1.1732146057455534e-05, "loss": 0.0009, "num_tokens": 3777860.0, "reward": 0.3552273988723755, "reward_std": 0.028055840730667116, "rewards/wrapper/mean": 0.17761370837688445, "rewards/wrapper/std": 0.019311992824077605, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 256.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 256.0, "completions/min_terminated_length": 0.0, "epoch": 1.975609756097561, "frac_reward_zero_std": 0.0, "grad_norm": 1.8470395803451538, "kl": 0.08914270466193557, "learning_rate": 3.884255434752975e-06, "loss": 0.0036, "num_tokens": 4317276.0, "reward": 0.3673787772655487, "reward_std": 0.028479888290166854, "rewards/wrapper/mean": 0.18368937373161315, "rewards/wrapper/std": 0.0193321343511343, "step": 40 }, { "epoch": 1.975609756097561, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 1.0, "eval_completions/max_length": 256.0, "eval_completions/max_terminated_length": 0.0, "eval_completions/mean_length": 256.0, "eval_completions/mean_terminated_length": 0.0, "eval_completions/min_length": 256.0, "eval_completions/min_terminated_length": 0.0, "eval_frac_reward_zero_std": 0.0, "eval_kl": 0.0393575007468462, "eval_loss": 0.0015887623885646462, "eval_num_tokens": 4317276.0, "eval_reward": 0.3702333927154541, "eval_reward_std": 0.026990585755556823, "eval_rewards/wrapper/mean": 0.18511669635772704, "eval_rewards/wrapper/std": 0.01845288172364235, "eval_runtime": 76.3221, "eval_samples_per_second": 2.62, "eval_steps_per_second": 0.17, "step": 40 } ], "logging_steps": 5, "max_steps": 40, "num_input_tokens_seen": 4317276, "num_train_epochs": 2, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 24, "trial_name": null, "trial_params": null }